{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":173125248,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{"_uuid":"f725a609-4b7e-4dfe-84b4-8edf158f0308","_cell_guid":"b5f1e48d-a434-43cd-b9c2-dc32a4ee4eca","trusted":true}},{"cell_type":"code","source":"import joblib\nfrom pathlib import Path\nimport gc\nfrom glob import glob\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'","metadata":{"_uuid":"72ffa5d5-199b-4771-a00b-78a16abe4d0a","_cell_guid":"9babf9a6-cfc0-4060-938b-c3d3b712da7c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:26:46.892715Z","iopub.execute_input":"2024-04-22T05:26:46.893198Z","iopub.status.idle":"2024-04-22T05:26:46.901405Z","shell.execute_reply.started":"2024-04-22T05:26:46.893161Z","shell.execute_reply":"2024-04-22T05:26:46.899061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n#เปลี่ยนรูปแบบข้อมูลเฉย\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n#เปลี่ยนรูปแบบข้อมูลวันที่\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n\n#รวมข้อมูล\nclass Aggregator:\n    # Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        # expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n\n        return expr_mean \n\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_first = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n\n        return expr_first\n\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        # expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        return expr_first # +expr_count\n    \n    def other_expr(df):\n        \n        numeric_cols = []\n        string_cols = []\n        \n        for col in df.columns:\n            \n            if df[col].dtype == pl.Float64:\n                numeric_cols.append(col)\n            elif df[col].dtype == pl.String:\n                string_cols.append(col)\n               \n\n        exprs = []\n\n        if numeric_cols:\n        # Use unique aliases for mean expressions\n            exprs += [pl.mean(col).alias(f\"mean_numeric_{col}\") for col in numeric_cols]\n\n        if string_cols:\n        # Use unique aliases for first expressions\n            exprs += [pl.first(col).alias(f\"first_string_{col}\") for col in string_cols]\n\n        return exprs\n\n\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        #expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        # expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        #expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return expr_first \n\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"_uuid":"d0f2ef1f-33e1-49d7-b779-583a224c71df","_cell_guid":"6891bc75-1da4-4719-95df-521340875c73","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:26:49.80759Z","iopub.execute_input":"2024-04-22T05:26:49.808003Z","iopub.status.idle":"2024-04-22T05:26:49.828078Z","shell.execute_reply.started":"2024-04-22T05:26:49.807972Z","shell.execute_reply":"2024-04-22T05:26:49.82679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''''#ดรอป80\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.8:\n                    df = df.drop(col)\n#ดรอปข้อมูลคอลัมน์ที่ row ซ้ำหมดเลย , ต่างกันมากเกิน 200 emcoding อาจไม่มีประสิทธิภาพมั้ง หาเพิ่มด่วน      \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_notebook_info = joblib.load('/kaggle/input/train-file-myversion/notebook_info.joblib')\nprint(f\"- [lgb] notebook_start_time: {lgb_notebook_info['notebook_start_time']}\")\nprint(f\"- [lgb] description: {lgb_notebook_info['description']}\")\n\ncols = lgb_notebook_info['cols']\n\nprint(f\"- [lgb] len(cols): {len(cols)}\")\n\n\nlgb_models = joblib.load('/kaggle/input/train-file-myversion/lgb_models.joblib')\nlgb_models","metadata":{"_uuid":"cf24fcbe-116b-4d4e-90b8-56230651c4d2","_cell_guid":"b98d7b92-e8f7-4a47-afa6-26824b6a9d07","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:26:56.380492Z","iopub.execute_input":"2024-04-22T05:26:56.380887Z","iopub.status.idle":"2024-04-22T05:26:56.756194Z","shell.execute_reply.started":"2024-04-22T05:26:56.380854Z","shell.execute_reply":"2024-04-22T05:26:56.754945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df  # Return the modified DataFrame\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    schema_lengths_before = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        print(f\"Loaded DataFrame from {path}: {df.shape}\")\n        \n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n            print(f\"Grouped and aggregated DataFrame for {path}: {df.shape}\")\n        \n        schema_lengths_before.append(len(df.schema))  # Store schema length before concatenation\n        chunks.append(df)\n    \n    # Print schema lengths before concatenation\n    print(\"Schema lengths before concatenation:\", schema_lengths_before)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    print(\"Concatenated DataFrame shape:\", df.shape)\n    \n    # Print schema length after concatenation\n    print(\"Schema length after concatenation:\", len(df.schema))\n    \n    df = df.unique(subset=[\"case_id\"])\n    print(\"Unique DataFrame shape:\", df.shape)\n    \n    return df  # Return the modified DataFrame","metadata":{"_uuid":"c7b68ba6-d3e6-4643-94bf-4c6a3f541f48","_cell_guid":"1bc52d09-8c21-4bdb-845c-79b8927ed3e9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:01.300712Z","iopub.execute_input":"2024-04-22T05:27:01.301082Z","iopub.status.idle":"2024-04-22T05:27:01.314576Z","shell.execute_reply.started":"2024-04-22T05:27:01.301056Z","shell.execute_reply":"2024-04-22T05:27:01.313371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base  # Return the modified DataFrame\n\n\n\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"_uuid":"1d5acef9-cef1-43eb-a1d5-615edb4c5520","_cell_guid":"11b792ab-b353-4eb8-9509-6b8229f56395","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:06.282855Z","iopub.execute_input":"2024-04-22T05:27:06.283337Z","iopub.status.idle":"2024-04-22T05:27:06.29484Z","shell.execute_reply.started":"2024-04-22T05:27:06.283301Z","shell.execute_reply":"2024-04-22T05:27:06.29327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"_uuid":"5157ae1f-a8fa-4a17-8e0e-a52d4bdad7c1","_cell_guid":"c50d4190-feff-49e0-a259-b4ace23b70c1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:07.95514Z","iopub.execute_input":"2024-04-22T05:27:07.95637Z","iopub.status.idle":"2024-04-22T05:27:07.967899Z","shell.execute_reply.started":"2024-04-22T05:27:07.956329Z","shell.execute_reply":"2024-04-22T05:27:07.966746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return expr_mean \n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_first = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        return expr_first\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return expr_first\n\n    @staticmethod\n    def other_expr(df):\n        numeric_cols = []\n        string_cols = []\n        for col in df.columns:\n            if df[col].dtype == pl.Float64:\n                numeric_cols.append(col)\n            elif df[col].dtype == pl.String:\n                string_cols.append(col)\n        exprs = []\n        if numeric_cols:\n            exprs += [pl.mean(col).alias(f\"mean_numeric_{col}\") for col in numeric_cols]\n        if string_cols:\n            exprs += [pl.first(col).alias(f\"first_string_{col}\") for col in string_cols]\n        return exprs\n\n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return expr_first \n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs\n\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    schema_lengths_before = []  # Store schema lengths before concatenation\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        # Print loaded DataFrame shape\n        print(f\"Loaded DataFrame from {path}: {df.shape}\")\n\n        # Store schema length before concatenation\n        schema_lengths_before.append(len(df.schema))\n\n        chunks.append(df)\n    \n    # Print schema lengths before concatenation\n    print(\"Schema lengths before concatenation:\", schema_lengths_before)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    \n    # Print schema length after concatenation\n    print(\"Schema length after concatenation:\", len(df.schema))\n    \n    df = df.unique(subset=[\"case_id\"])\n    return df\n\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}","metadata":{"_uuid":"27efe051-4858-48ab-8931-88c7fd624b64","_cell_guid":"a088ee65-362e-410b-9d29-892e9cb267fe","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:10.891446Z","iopub.execute_input":"2024-04-22T05:27:10.891814Z","iopub.status.idle":"2024-04-22T05:27:11.09002Z","shell.execute_reply.started":"2024-04-22T05:27:10.891783Z","shell.execute_reply":"2024-04-22T05:27:11.088696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)\ndel data_store\n#df_test = df_test.pipe(Pipeline.filter_cols)\ngc.collect()\n\n#df_test = df_test.select(['case_id'] + cols)\n\n\n#df_test = reduce_mem_usage(df_test)\n#df_test = df_test.set_index('case_id')\n#print(\"test data shape:\\t\", df_test.shape)\n\n#gc.collect()","metadata":{"_uuid":"5037f114-0214-4081-bda0-95271cddec45","_cell_guid":"df9547f1-242d-4d19-a10f-bfaf412f6cdd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:36.822701Z","iopub.execute_input":"2024-04-22T05:27:36.823102Z","iopub.status.idle":"2024-04-22T05:27:37.357066Z","shell.execute_reply.started":"2024-04-22T05:27:36.823072Z","shell.execute_reply":"2024-04-22T05:27:37.3559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"_uuid":"77ae64e8-fd50-4d7c-8fbe-a6ddc530d528","_cell_guid":"c11d3cdc-78dd-44bc-9227-eac86c124643","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:40.669538Z","iopub.execute_input":"2024-04-22T05:27:40.669935Z","iopub.status.idle":"2024-04-22T05:27:40.692749Z","shell.execute_reply.started":"2024-04-22T05:27:40.669904Z","shell.execute_reply":"2024-04-22T05:27:40.691365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Convert Polars DataFrame to pandas DataFrame\ndf_test_pandas = df_test.to_pandas()\n\n# Now df_train_pandas is a pandas DataFrame\nprint(\"Type of df_train_pandas:\", type(df_test_pandas))","metadata":{"_uuid":"cc0175d1-ffda-4fbd-b330-b6cd6f0b6b0a","_cell_guid":"3a5bdc83-a303-4eb7-a083-74536bbf68d5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:27:45.851092Z","iopub.execute_input":"2024-04-22T05:27:45.851469Z","iopub.status.idle":"2024-04-22T05:27:45.874769Z","shell.execute_reply.started":"2024-04-22T05:27:45.85144Z","shell.execute_reply":"2024-04-22T05:27:45.873246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf_test_pandas","metadata":{"_uuid":"5d423554-3248-44bc-abb5-467246d52b27","_cell_guid":"d7d49371-b1c2-4715-bb24-17103884b1ad","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:12.37571Z","iopub.execute_input":"2024-04-22T05:28:12.376669Z","iopub.status.idle":"2024-04-22T05:28:12.871578Z","shell.execute_reply.started":"2024-04-22T05:28:12.376624Z","shell.execute_reply":"2024-04-22T05:28:12.870466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndef fill_nan_values(df):\n    # Iterate through each column in the DataFrame\n    for column in df.columns:\n        # Get the data type of the column\n        column_dtype = df[column].dtype\n        \n        if column_dtype == 'float64' or column_dtype == 'int64':\n            # If the column is of float or integer data type, fill NaN with the mean\n            mean_value = df[column].mean()\n            df[column].fillna(mean_value, inplace=True)\n        elif column_dtype == 'category' or column_dtype == 'object' or column_dtype == 'bool':\n            # If the column is of categorical, object (string), or boolean data type, fill NaN with the mode\n            mode_value = df[column].mode()\n            if not mode_value.empty:\n                df[column].fillna(mode_value.iloc[0], inplace=True)\n                \n    return df\n\n# Sample usage:\n# df = pd.read_csv('your_data.csv')\n# filled_df = fill_nan_values(df)","metadata":{"_uuid":"cb77701a-50d5-4bd0-9f1a-5a8e9dc13b02","_cell_guid":"fc137637-6c95-4caf-8a35-33fdb90cfb09","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:22.191039Z","iopub.execute_input":"2024-04-22T05:28:22.19147Z","iopub.status.idle":"2024-04-22T05:28:22.199571Z","shell.execute_reply.started":"2024-04-22T05:28:22.191436Z","shell.execute_reply":"2024-04-22T05:28:22.198408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train = fill_nan_values(df_test_pandas)","metadata":{"_uuid":"cec9ec61-f1d5-4a82-a685-82aa2b644b0d","_cell_guid":"100873b3-e1cd-4489-a38e-5dbe7217d5a2","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:24.140372Z","iopub.execute_input":"2024-04-22T05:28:24.140766Z","iopub.status.idle":"2024-04-22T05:28:24.343568Z","shell.execute_reply.started":"2024-04-22T05:28:24.140733Z","shell.execute_reply":"2024-04-22T05:28:24.342338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train","metadata":{"_uuid":"b4cfef2e-dee8-4f19-b6b4-b8fa3d2507b5","_cell_guid":"cdb5d80d-9a3a-4f84-993a-b48d65f902e8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:27.221315Z","iopub.execute_input":"2024-04-22T05:28:27.222255Z","iopub.status.idle":"2024-04-22T05:28:27.731444Z","shell.execute_reply.started":"2024-04-22T05:28:27.222221Z","shell.execute_reply":"2024-04-22T05:28:27.73038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\ndef label_encode_categorical(df):\n    # Initialize a LabelEncoder object\n    label_encoder = LabelEncoder()\n    \n    # Iterate through each column in the DataFrame\n    for column in df.columns:\n        # Get the data type of the column\n        column_dtype = df[column].dtype\n        \n        if column_dtype == 'category' or column_dtype == 'object' or column_dtype == 'bool':\n            # If the column is of categorical, object (string), or boolean data type, perform label encoding\n            df[column] = label_encoder.fit_transform(df[column])\n                \n    return df\n\n# Sample usage:\n# df = pd.read_csv('your_data.csv')\n# encoded_df = label_encode_categorical(df)","metadata":{"_uuid":"a4ff2980-0d47-47ce-b425-1d9373e40507","_cell_guid":"c3b9da64-38c0-4cab-887d-072aeb5bc349","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:32.959116Z","iopub.execute_input":"2024-04-22T05:28:32.960429Z","iopub.status.idle":"2024-04-22T05:28:32.972192Z","shell.execute_reply.started":"2024-04-22T05:28:32.960374Z","shell.execute_reply":"2024-04-22T05:28:32.970921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train = label_encode_categorical(new_train)\npd.set_option('display.max_columns', None)\nnew_train","metadata":{"_uuid":"7ef908e8-dd52-4528-a53a-75c30aef8c44","_cell_guid":"b7c9c264-f160-41f1-b7e5-04fc29788934","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:28:35.606869Z","iopub.execute_input":"2024-04-22T05:28:35.607729Z","iopub.status.idle":"2024-04-22T05:28:36.199081Z","shell.execute_reply.started":"2024-04-22T05:28:35.607692Z","shell.execute_reply":"2024-04-22T05:28:36.198036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the columns to drop\ncolumns_to_drop = [col for col in new_train.columns if col.startswith('first_num_group')]\n\n# Drop the columns\nnew_train = new_train.drop(columns_to_drop, axis=1)\npd.set_option('display.max_columns', None)\nnew_train.shape","metadata":{"_uuid":"37b8ddb4-9166-4b71-be60-1c2a306aebe6","_cell_guid":"7b3ae02f-80f1-42a5-bb56-0028fbdbfb2f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-22T05:41:26.516265Z","iopub.execute_input":"2024-04-22T05:41:26.516722Z","iopub.status.idle":"2024-04-22T05:41:26.533866Z","shell.execute_reply.started":"2024-04-22T05:41:26.516687Z","shell.execute_reply":"2024-04-22T05:41:26.532609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 =pd.read_csv('/kaggle/input/train-file-myversion/train_base.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf2","metadata":{"execution":{"iopub.status.busy":"2024-04-22T05:55:45.867748Z","iopub.execute_input":"2024-04-22T05:55:45.868601Z","iopub.status.idle":"2024-04-22T05:55:46.230323Z","shell.execute_reply.started":"2024-04-22T05:55:45.868566Z","shell.execute_reply":"2024-04-22T05:55:46.228862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndef drop_matching_columns(df1, df2):\n    columns_to_drop = [col for col in df2.columns if col in df1.columns]\n    df2.drop(columns=columns_to_drop, inplace=True)\n    return df2\n\n# Example usage:\n# Assuming df1 and df2 are your DataFrames\n\n\ndf2 = drop_matching_columns(new_train, df2)\n\nprint(\"\\nAfter dropping columns in df2:\")\ndf2\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T05:59:37.19442Z","iopub.execute_input":"2024-04-22T05:59:37.19549Z","iopub.status.idle":"2024-04-22T05:59:39.571661Z","shell.execute_reply.started":"2024-04-22T05:59:37.195441Z","shell.execute_reply":"2024-04-22T05:59:39.570347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drop_columns_not_matching(df1, df2):\n    columns_to_drop = [col for col in df1.columns if col not in df2.columns]\n    df1.drop(columns=columns_to_drop, inplace=True)\n    return df1\n\ndf1 = drop_columns_not_matching(new_train, df2)\n\nprint(\"\\nAfter dropping columns:\")\ndf1.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T06:00:45.086013Z","iopub.execute_input":"2024-04-22T06:00:45.086449Z","iopub.status.idle":"2024-04-22T06:00:45.098625Z","shell.execute_reply.started":"2024-04-22T06:00:45.086419Z","shell.execute_reply":"2024-04-22T06:00:45.097229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = reduce_mem_usage(new_train)","metadata":{"_uuid":"7b6647e6-3a56-436f-8189-232ebfd721bb","_cell_guid":"d2ffc97f-47f7-4b9f-bc7d-f16d9a9adf29","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n     \n    def predict_proba(self, X):      \n        # lgb\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators[:5]]\n        \n        # cat        \n        X[cat_cols] = X[cat_cols].astype(str)\n        y_preds += [estimator.predict_proba(X) for estimator in self.estimators[-5:]]\n        \n        return np.mean(y_preds, axis=0)","metadata":{"_uuid":"db7d3824-cb39-4fa8-ab64-8bd12afdc61a","_cell_guid":"45a2ad51-1e93-43f5-846f-d26d9128348c","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = VotingModel(lgb_models )\nlen(model.estimators)","metadata":{"_uuid":"783b2d9f-d840-4d1e-a6ef-0331261c260a","_cell_guid":"2572d1bf-1f31-40ec-8cb4-60f3c0d1a5f9","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.Series(model.predict_proba(df_test)[:, 1], index=df_test.index)\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred\ndf_subm.to_csv(\"submission.csv\")\ndf_subm","metadata":{"_uuid":"6f786879-0a45-4231-8250-d841fca7ede4","_cell_guid":"6d76d7a5-5147-4531-945c-0d1ca61c33b7","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}