{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8207910,"sourceType":"datasetVersion","datasetId":4863667}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings,os,glob,gc,json\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-04T16:09:27.183816Z","iopub.execute_input":"2024-05-04T16:09:27.184270Z","iopub.status.idle":"2024-05-04T16:09:30.408575Z","shell.execute_reply.started":"2024-05-04T16:09:27.184234Z","shell.execute_reply":"2024-05-04T16:09:30.407231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"box-shadow: rgba(0, 0, 0, 0.16) 0px 1px 4px inset, rgb(51, 51, 51) 0px 0px 0px 3px inset; padding:20px; font-size:32px; font-family: consolas; text-align:center; display:fill; border-radius:15px; color:rgb(34, 34, 34);　background-color:rgb(255,255,255); \"> <b> Data Loading and Preprocessing functions </b></div>","metadata":{}},{"cell_type":"code","source":"class wrangling:\n\n    @staticmethod\n    def set_datatypes(df):\n        \"\"\"\n        ends with p,a float, m string,D date\n        \"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\" as int\n        date_decision as date\n        \"\"\"\n        for col in df.columns:\n            if col in (\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"):\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col[-1] in ('D',) or col in (\"date_decision\"):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\",\"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n    \n     \n    @staticmethod\n    def filter_cols(df):\n        \n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:32.068409Z","iopub.execute_input":"2024-05-04T16:09:32.069100Z","iopub.status.idle":"2024-05-04T16:09:32.091153Z","shell.execute_reply.started":"2024-05-04T16:09:32.069058Z","shell.execute_reply":"2024-05-04T16:09:32.089336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    \"\"\"\n        ends with p,a float, m string,D date\n        \"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\" as int\n        date_decision as date\n    \"\"\"\n    \n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\",\"A\")]\n        expr = []\n        expr.extend([pl.max(col).alias(f'max_{col}') for col in cols])\n        expr.extend([pl.min(col).alias(f'min_{col}') for col in cols])\n        expr.extend([pl.mean(col).alias(f'mean_{col}') for col in cols])\n        return expr\n    \n    @staticmethod\n    def date_expr(df):\n        expr = []\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                expr.append((pl.col(\"date_decision\") - pl.col(col)).dt.year().alias(f\"year_gap_{col}\"))\n                expr.append((pl.col(\"date_decision\") - pl.col(col)).dt.month().alias(f\"month_gap_{col}\"))\n                expr.append((pl.col(\"date_decision\") - pl.col(col)).dt.days().alias(f\"days_gap_{col}\"))\n        return expr\n    \n    @staticmethod\n    def string_expr(df):\n        expr = []\n        for col in df.columns:\n            if col[-1] in (\"M\",):\n                expr.append(pl.col(col).str.strip().str.len_chars().mean().alias(f'text_length_{col}'))\n        return expr\n    \n    @staticmethod\n    def other_expr(df):\n        expr = []\n        for col in df.columns:\n            if col[-1] in (\"T\", \"L\"):\n                expr.append(pl.max(col).alias(f'max_{col}'))\n                expr.append(pl.min(col).alias(f'min_{col}'))\n        return expr\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.string_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:32.550676Z","iopub.execute_input":"2024-05-04T16:09:32.551289Z","iopub.status.idle":"2024-05-04T16:09:32.577630Z","shell.execute_reply.started":"2024-05-04T16:09:32.551241Z","shell.execute_reply":"2024-05-04T16:09:32.576137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def topandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data,cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:32.953407Z","iopub.execute_input":"2024-05-04T16:09:32.954455Z","iopub.status.idle":"2024-05-04T16:09:32.962375Z","shell.execute_reply.started":"2024-05-04T16:09:32.954402Z","shell.execute_reply":"2024-05-04T16:09:32.961094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\ndef reduce_mem_usage(df, float16_as32=True):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)                    \n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:33.354303Z","iopub.execute_input":"2024-05-04T16:09:33.354827Z","iopub.status.idle":"2024-05-04T16:09:33.372260Z","shell.execute_reply.started":"2024-05-04T16:09:33.354789Z","shell.execute_reply":"2024-05-04T16:09:33.370822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"box-shadow: rgba(0, 0, 0, 0.16) 0px 1px 4px inset, rgb(51, 51, 51) 0px 0px 0px 3px inset; padding:20px; font-size:32px; font-family: consolas; text-align:center; display:fill; border-radius:15px; color:rgb(34, 34, 34);　background-color:rgb(255,255,255); \"> <b> Loading Data </b></div>","metadata":{}},{"cell_type":"code","source":"root = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/\"\nTRAIN_DIR = root + \"train/\"\nTEST_DIR = root + \"test/\"\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        print(df_base.shape)\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\",suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(wrangling.handle_dates)\n    df_base = df_base.unique(subset=[\"case_id\"])\n    return df_base\n\ndef read_file(path,non_imp_columns,depth=None):\n    df = pl.read_parquet(path).pipe(wrangling.set_datatypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    df = df.unique(subset=[\"case_id\"])\n    df = df.pipe(wrangling.filter_cols)\n    return df[list(set(df.columns)-set(non_imp_columns))]\n\ndef read_files(regex_path,non_imp_columns, depth=None):\n    chunks = []\n    for path in glob.glob(str(regex_path)):\n        df = pl.read_parquet(path).pipe(wrangling.set_datatypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    df = df.pipe(wrangling.filter_cols)\n    \n    return df[list(set(df.columns)-set(non_imp_columns))]","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:34.318274Z","iopub.execute_input":"2024-05-04T16:09:34.318653Z","iopub.status.idle":"2024-05-04T16:09:34.335169Z","shell.execute_reply.started":"2024-05-04T16:09:34.318624Z","shell.execute_reply":"2024-05-04T16:09:34.333517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith open('/kaggle/input/credit-risk-non-imp-columns/my_data.json', 'r') as f:\n    json_string = f.read()\n    data = json.loads(json_string)\ndata_store = {\n    \"df_base\": read_file(TRAIN_DIR + \"train_base.parquet\",data[\"non_imp_columns\"]),\n    \"depth_0\": [\n        read_file(TRAIN_DIR + \"train_static_cb_0.parquet\",data[\"non_imp_columns\"]),\n        read_files(TRAIN_DIR + \"train_static_0_*.parquet\",data[\"non_imp_columns\"]),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR + \"train_applprev_1_*.parquet\",data[\"non_imp_columns\"], 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_a_1.parquet\",data[\"non_imp_columns\"], 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_b_1.parquet\",data[\"non_imp_columns\"], 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_c_1.parquet\",data[\"non_imp_columns\"], 1),\n        read_files(TRAIN_DIR + \"train_credit_bureau_a_1_*.parquet\", data[\"non_imp_columns\"],1),\n        read_file(TRAIN_DIR + \"train_credit_bureau_b_1.parquet\", data[\"non_imp_columns\"],1),\n        read_file(TRAIN_DIR + \"train_other_1.parquet\",data[\"non_imp_columns\"], 1),\n        read_file(TRAIN_DIR + \"train_person_1.parquet\",data[\"non_imp_columns\"], 1),\n        read_file(TRAIN_DIR + \"train_deposit_1.parquet\", data[\"non_imp_columns\"],1),\n        read_file(TRAIN_DIR + \"train_debitcard_1.parquet\", data[\"non_imp_columns\"],1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR + \"train_credit_bureau_b_2.parquet\",data[\"non_imp_columns\"], 2),\n        read_files(TRAIN_DIR + \"train_credit_bureau_a_2_*.parquet\",data[\"non_imp_columns\"], 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:09:34.860850Z","iopub.execute_input":"2024-05-04T16:09:34.861244Z","iopub.status.idle":"2024-05-04T16:13:41.840423Z","shell.execute_reply.started":"2024-05-04T16:09:34.861214Z","shell.execute_reply":"2024-05-04T16:13:41.838914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)\n\ndel data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:13:41.843901Z","iopub.execute_input":"2024-05-04T16:13:41.844372Z","iopub.status.idle":"2024-05-04T16:13:49.155450Z","shell.execute_reply.started":"2024-05-04T16:13:41.844317Z","shell.execute_reply":"2024-05-04T16:13:49.154283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train,cat_cols = topandas(df_train)\n\ndf_train = reduce_mem_usage(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:13:49.157336Z","iopub.execute_input":"2024-05-04T16:13:49.158080Z","iopub.status.idle":"2024-05-04T16:13:57.044267Z","shell.execute_reply.started":"2024-05-04T16:13:49.158039Z","shell.execute_reply":"2024-05-04T16:13:57.042803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.index = df_train[\"case_id\"]\ndf_train.drop(['case_id',\"target\", 'WEEK_NUM',\"month_decision\",\"weekday_decision\",\"birthdate_574D\",\"dateofbirth_337D\"], axis=1,inplace=True)\nprint(gc.collect())","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:13:57.047006Z","iopub.execute_input":"2024-05-04T16:13:57.047356Z","iopub.status.idle":"2024-05-04T16:13:58.511057Z","shell.execute_reply.started":"2024-05-04T16:13:57.047326Z","shell.execute_reply":"2024-05-04T16:13:58.509875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"box-shadow: rgba(0, 0, 0, 0.16) 0px 1px 4px inset, rgb(51, 51, 51) 0px 0px 0px 3px inset; padding:20px; font-size:32px; font-family: consolas; text-align:center; display:fill; border-radius:15px; color:rgb(34, 34, 34);　background-color:rgb(255,255,255); \"> <b> Null Value Imputation </b></div>","metadata":{}},{"cell_type":"code","source":"fill_cat_values = {}\nfor col in cat_cols:\n    cats = df_train[col].cat.categories\n    fill_cat_values[col] = dict(zip(cats,range(len(cats))))\n    df_train[col].replace(fill_cat_values[col],inplace=True)\n    if df_train[col].isnull().sum() > 0:\n        df_train[col] = df_train[col].cat.add_categories([-1]).fillna(-1)\ndf_train.shape,len(fill_cat_values),len(cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:13:58.512470Z","iopub.execute_input":"2024-05-04T16:13:58.512837Z","iopub.status.idle":"2024-05-04T16:13:58.666001Z","shell.execute_reply.started":"2024-05-04T16:13:58.512806Z","shell.execute_reply":"2024-05-04T16:13:58.664765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_cat_columns = list(set(df_train.columns)-set(cat_cols))\nfill_num_values = df_train[non_cat_columns].mode().iloc[0]\ndf_train.fillna(fill_num_values,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:13:58.668139Z","iopub.execute_input":"2024-05-04T16:13:58.668555Z","iopub.status.idle":"2024-05-04T16:14:07.569170Z","shell.execute_reply.started":"2024-05-04T16:13:58.668522Z","shell.execute_reply":"2024-05-04T16:14:07.568031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del cat_cols,non_cat_columns,fill_cat_values,fill_num_values\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:14:07.570597Z","iopub.execute_input":"2024-05-04T16:14:07.570957Z","iopub.status.idle":"2024-05-04T16:14:07.578290Z","shell.execute_reply.started":"2024-05-04T16:14:07.570926Z","shell.execute_reply":"2024-05-04T16:14:07.577220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_train = pd.read_parquet(TRAIN_DIR + \"train_base.parquet\",columns=[\"case_id\",\"WEEK_NUM\",\"target\"])\nbase_train.index = base_train[\"case_id\"]\nbase_train.drop(\"case_id\",axis=1,inplace=True)\ny_train = base_train.loc[df_train.index,\"target\"]\ny_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:14:07.579829Z","iopub.execute_input":"2024-05-04T16:14:07.580212Z","iopub.status.idle":"2024-05-04T16:14:08.038237Z","shell.execute_reply.started":"2024-05-04T16:14:07.580177Z","shell.execute_reply":"2024-05-04T16:14:08.037096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evalvation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score,precision_score\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, yy_train, y_test = train_test_split(df_train, y_train, test_size=0.2, random_state=42)\n\ndef evalvate(model,X,y):\n    y_pred = model.predict(X)\n    print(f\"ROC AUC Score : {roc_auc_score(y,y_pred)}\")\n    y_pred = np.where(y_pred>=0.5,1,0)\n    print(f\"Precision Score : {precision_score(y,y_pred)}\")\n    print(\"*\"*50)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:14:08.039886Z","iopub.execute_input":"2024-05-04T16:14:08.040195Z","iopub.status.idle":"2024-05-04T16:14:10.972921Z","shell.execute_reply.started":"2024-05-04T16:14:08.040168Z","shell.execute_reply":"2024-05-04T16:14:10.971522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"box-shadow: rgba(0, 0, 0, 0.16) 0px 1px 4px inset, rgb(51, 51, 51) 0px 0px 0px 3px inset; padding:20px; font-size:32px; font-family: consolas; text-align:center; display:fill; border-radius:15px; color:rgb(34, 34, 34);　background-color:rgb(255,255,255); \"> <b> Model Sections by evalvate on Data </b></div>","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlr = LogisticRegression()\nlr.fit(df_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T11:57:14.189283Z","iopub.execute_input":"2024-05-03T11:57:14.189801Z","iopub.status.idle":"2024-05-03T11:59:13.468242Z","shell.execute_reply.started":"2024-05-03T11:57:14.189767Z","shell.execute_reply":"2024-05-03T11:59:13.466548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(lr,df_train,y_train)\nevalvate(lr,X_train,yy_train)\nevalvate(lr,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:01:12.666891Z","iopub.execute_input":"2024-05-03T12:01:12.667356Z","iopub.status.idle":"2024-05-03T12:01:17.038420Z","shell.execute_reply.started":"2024-05-03T12:01:12.667325Z","shell.execute_reply":"2024-05-03T12:01:17.037098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(lr.coef_.reshape(-1),index=df_train.columns).sort_values(ascending=False)[:3].plot(kind=\"barh\")","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:02:43.398061Z","iopub.execute_input":"2024-04-30T12:02:43.402774Z","iopub.status.idle":"2024-04-30T12:02:43.871723Z","shell.execute_reply.started":"2024-04-30T12:02:43.402702Z","shell.execute_reply":"2024-04-30T12:02:43.870036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(lr.coef_.reshape(-1),index=df_train.columns).sort_values(ascending=True)[:3].plot(kind=\"barh\")","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:02:43.874830Z","iopub.execute_input":"2024-04-30T12:02:43.875272Z","iopub.status.idle":"2024-04-30T12:02:44.286948Z","shell.execute_reply.started":"2024-04-30T12:02:43.875241Z","shell.execute_reply":"2024-04-30T12:02:44.285895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndc = DecisionTreeClassifier(max_depth=4,max_leaf_nodes=10)\ndc.fit(df_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:03:28.512672Z","iopub.execute_input":"2024-05-03T12:03:28.513275Z","iopub.status.idle":"2024-05-03T12:05:06.445605Z","shell.execute_reply.started":"2024-05-03T12:03:28.513240Z","shell.execute_reply":"2024-05-03T12:05:06.443871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(dc,df_train,y_train)\nevalvate(dc,X_train,yy_train)\nevalvate(dc,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T12:05:06.448155Z","iopub.execute_input":"2024-05-03T12:05:06.448604Z","iopub.status.idle":"2024-05-03T12:05:11.751762Z","shell.execute_reply.started":"2024-05-03T12:05:06.448568Z","shell.execute_reply":"2024-05-03T12:05:11.750049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(dc.feature_importances_,index=df_train.columns).sort_values(ascending=False)[:3].plot(kind=\"barh\")","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:04:00.860541Z","iopub.execute_input":"2024-04-30T12:04:00.861655Z","iopub.status.idle":"2024-04-30T12:04:01.419268Z","shell.execute_reply.started":"2024-04-30T12:04:00.861609Z","shell.execute_reply":"2024-04-30T12:04:01.417909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary Both Logistic regression & Decision Tree is essentially making random guesses.","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"box-shadow: rgba(0, 0, 0, 0.16) 0px 1px 4px inset, rgb(51, 51, 51) 0px 0px 0px 3px inset; padding:20px; font-size:32px; font-family: consolas; text-align:center; display:fill; border-radius:15px; color:rgb(34, 34, 34);　background-color:rgb(255,255,255); \"> <b> Light GBM Model </b></div>\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport lightgbm as lgb\n\nlgb_train = lgb.Dataset(X_train, label=yy_train)\nlgb_valid = lgb.Dataset(X_test, label=y_test, reference=lgb_train)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\n#gbm = lgb.LGBMClassifier(**params)\n\n#gbm.fit(X=X_train,y=yy_train,eval_set=[(X_test,y_test)],callbacks=[lgb.log_evaluation(20)])\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(20), lgb.early_stopping(5)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:56:18.620621Z","iopub.execute_input":"2024-05-04T16:56:18.621916Z","iopub.status.idle":"2024-05-04T16:59:08.771860Z","shell.execute_reply.started":"2024-05-04T16:56:18.621875Z","shell.execute_reply":"2024-05-04T16:59:08.770121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def render_plot_importance(importance_type, max_features=10, ignore_zero=True, precision=3):\n    lgb.plot_importance(\n        gbm,\n        importance_type=importance_type,\n        max_num_features=max_features,\n        ignore_zero=ignore_zero,\n        figsize=(12, 8),\n        precision=precision,\n    )\n    plt.show()\nrender_plot_importance(importance_type=\"gain\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:33.718136Z","iopub.execute_input":"2024-05-04T16:19:33.719950Z","iopub.status.idle":"2024-05-04T16:19:34.192661Z","shell.execute_reply.started":"2024-05-04T16:19:33.719898Z","shell.execute_reply":"2024-05-04T16:19:34.191017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(gbm,df_train,y_train)\nevalvate(gbm,X_train,yy_train) \nevalvate(gbm,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T17:00:27.254093Z","iopub.execute_input":"2024-05-04T17:00:27.254604Z","iopub.status.idle":"2024-05-04T17:01:08.809285Z","shell.execute_reply.started":"2024-05-04T17:00:27.254565Z","shell.execute_reply":"2024-05-04T17:01:08.807804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Catbooting model","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier\n\ncat = CatBoostClassifier(eval_metric=\"AUC\",learning_rate=0.001,cat_features=X_train.select_dtypes(\"category\").columns.to_list(),\n                         early_stopping_rounds=10)\n\ncat.fit(X=X_train,y=yy_train,eval_set=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2024-05-03T11:16:06.378421Z","iopub.execute_input":"2024-05-03T11:16:06.378896Z","iopub.status.idle":"2024-05-03T11:17:36.484396Z","shell.execute_reply.started":"2024-05-03T11:16:06.378854Z","shell.execute_reply":"2024-05-03T11:17:36.483047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(cat,df_train,y_train)\nevalvate(cat,X_train,yy_train)\nevalvate(cat,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T11:18:42.236020Z","iopub.execute_input":"2024-05-03T11:18:42.236535Z","iopub.status.idle":"2024-05-03T11:18:48.493126Z","shell.execute_reply.started":"2024-05-03T11:18:42.236493Z","shell.execute_reply":"2024-05-03T11:18:48.491435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Nural network","metadata":{}},{"cell_type":"code","source":"from sklearn.neural_network import MLPClassifier\nmlp = MLPClassifier(hidden_layer_sizes=(100, 50), activation=\"logistic\", alpha=0.01)\nmlp.fit(X_train, yy_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T11:30:43.904773Z","iopub.execute_input":"2024-05-03T11:30:43.905265Z","iopub.status.idle":"2024-05-03T11:38:43.223364Z","shell.execute_reply.started":"2024-05-03T11:30:43.905230Z","shell.execute_reply":"2024-05-03T11:38:43.221344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(mlp,df_train,y_train)\nevalvate(mlp,X_train,yy_train)\nevalvate(mlp,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-03T11:46:14.918894Z","iopub.execute_input":"2024-05-03T11:46:14.922109Z","iopub.status.idle":"2024-05-03T11:46:34.205934Z","shell.execute_reply.started":"2024-05-03T11:46:14.922062Z","shell.execute_reply":"2024-05-03T11:46:34.204723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One Class SVM (support vector machine)","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import OneClassSVM\n\n# nu is the parameter to control the fraction of outliers\nocsvm = OneClassSVM(kernel='rbf', nu=0.1)\nocsvm.fit(X_test.iloc[:1000], y_test.iloc[:1000])","metadata":{"execution":{"iopub.status.busy":"2024-05-04T07:26:18.025188Z","iopub.execute_input":"2024-05-04T07:26:18.025642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(ocsvm,df_train,y_train)\nevalvate(ocsvm,X_train,yy_train)\nevalvate(ocsvm,X_test,y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IsolationForest Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import IsolationForest\n\n# contamination is the expected proportion of outliers\nisolation_forest = IsolationForest(contamination=0.1)  \nisolation_forest.fit(X_train, yy_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T07:19:56.310206Z","iopub.execute_input":"2024-05-04T07:19:56.310653Z","iopub.status.idle":"2024-05-04T07:20:43.400810Z","shell.execute_reply.started":"2024-05-04T07:19:56.310617Z","shell.execute_reply":"2024-05-04T07:20:43.399542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(isolation_forest,df_train,y_train)\nevalvate(isolation_forest,X_train,yy_train)\nevalvate(isolation_forest,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T07:20:43.402476Z","iopub.execute_input":"2024-05-04T07:20:43.402878Z","iopub.status.idle":"2024-05-04T07:25:05.371771Z","shell.execute_reply.started":"2024-05-04T07:20:43.402847Z","shell.execute_reply":"2024-05-04T07:25:05.370340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XG Boosting Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.calibration import CalibratedClassifierCV\nfrom xgboost import XGBClassifier\n\nxgb = XGBClassifier(objective='binary:logistic',enable_categorical=True,n_estimators=100)\n\ncc = CalibratedClassifierCV(xgb,cv=2,method=\"sigmoid\")\n\ncc.fit(X_train,yy_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T15:10:18.081667Z","iopub.execute_input":"2024-05-04T15:10:18.082169Z","iopub.status.idle":"2024-05-04T15:11:36.130774Z","shell.execute_reply.started":"2024-05-04T15:10:18.082107Z","shell.execute_reply":"2024-05-04T15:11:36.129414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalvate(cc,df_train,y_train)\nevalvate(cc,X_train,yy_train)\nevalvate(cc,X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T15:11:36.132953Z","iopub.execute_input":"2024-05-04T15:11:36.133344Z","iopub.status.idle":"2024-05-04T15:11:52.955667Z","shell.execute_reply.started":"2024-05-04T15:11:36.133307Z","shell.execute_reply":"2024-05-04T15:11:52.954139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lime import lime_tabular\nexplainer = lime_tabular.LimeTabularExplainer(df_train.values,training_labels=y_train.values, feature_names=df_train.columns)\nfor i in y_train[y_train==1].index[:5]:\n    exp = explainer.explain_instance(df_train.loc[i].values, cc.predict_proba, num_features=5, top_labels=1)\n    exp.show_in_notebook(show_table=True, show_all=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T15:11:52.957289Z","iopub.execute_input":"2024-05-04T15:11:52.957729Z","iopub.status.idle":"2024-05-04T15:14:07.201504Z","shell.execute_reply.started":"2024-05-04T15:11:52.957695Z","shell.execute_reply":"2024-05-04T15:14:07.199269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary Catbooting , Nural Network , One Class SVM , Isolation Forest is essentially making random guesses.\n\n# LightGBM model > XGbooting models are performing Normal","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}