{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-02T16:16:06.730322Z","iopub.execute_input":"2024-06-02T16:16:06.730873Z","iopub.status.idle":"2024-06-02T16:16:07.358114Z","shell.execute_reply.started":"2024-06-02T16:16:06.730819Z","shell.execute_reply":"2024-06-02T16:16:07.356475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport sys\nimport psutil\nfrom datetime import datetime","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:07.360604Z","iopub.execute_input":"2024-06-02T16:16:07.361256Z","iopub.status.idle":"2024-06-02T16:16:07.367822Z","shell.execute_reply.started":"2024-06-02T16:16:07.361220Z","shell.execute_reply":"2024-06-02T16:16:07.366274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install polars==0.20.26\nimport polars as pl\npl.show_versions()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:07.369948Z","iopub.execute_input":"2024-06-02T16:16:07.370474Z","iopub.status.idle":"2024-06-02T16:16:09.043433Z","shell.execute_reply.started":"2024-06-02T16:16:07.370432Z","shell.execute_reply":"2024-06-02T16:16:09.042312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/intermediate\n\n!mkdir /kaggle/working/intermediate/train/\n!mkdir /kaggle/working/intermediate/test/\n\n!mkdir /kaggle/working/intermediate/train/base\n!mkdir /kaggle/working/intermediate/test/base\n\n!mkdir /kaggle/working/intermediate/train/applprev_1\n!mkdir /kaggle/working/intermediate/test/applprev_1\n\n!mkdir /kaggle/working/intermediate/train/other_1\n!mkdir /kaggle/working/intermediate/test/other_1\n\n!mkdir /kaggle/working/intermediate/train/tax_registry\n!mkdir /kaggle/working/intermediate/test/tax_registry","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:09.047118Z","iopub.execute_input":"2024-06-02T16:16:09.047694Z","iopub.status.idle":"2024-06-02T16:16:21.261865Z","shell.execute_reply.started":"2024-06-02T16:16:09.047658Z","shell.execute_reply":"2024-06-02T16:16:21.260403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_PATH = \"/kaggle/working/intermediate\"\nDATA_PATH = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files\"","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:21.263765Z","iopub.execute_input":"2024-06-02T16:16:21.264161Z","iopub.status.idle":"2024-06-02T16:16:21.270224Z","shell.execute_reply.started":"2024-06-02T16:16:21.264124Z","shell.execute_reply":"2024-06-02T16:16:21.268888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def return_test_submission():\n    test_base = pd.read_parquet(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/test_base.parquet\")\n    pred = []\n    for i in range(test_base.shape[0]):\n        pred.append(abs(np.random.randn()))\n    score = pd.DataFrame({\"score\": pred})\n    submission = pd.concat([test_base[[\"case_id\",]], score], axis=1)\n    submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:21.271572Z","iopub.execute_input":"2024-06-02T16:16:21.271895Z","iopub.status.idle":"2024-06-02T16:16:21.285197Z","shell.execute_reply.started":"2024-06-02T16:16:21.271867Z","shell.execute_reply":"2024-06-02T16:16:21.283832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_columns(df, threshold, columns): #threshold (0, 1) percentage\n    arr = [col for col in df.columns if df.select( pl.col(col).is_null().sum()/df[col].shape[0] < threshold ).row(0)[0]]\n    return arr\n\ndef fill_na(df):\n    return (df\n            .fill_null(strategy=\"forward\")\n            .fill_null(strategy=\"backward\")\n            .select(pl.col(\"case_id\"),\n                    pl.col(pl.Float64, pl.Int64).exclude(\"case_id\").fill_nan('mean').cast(pl.Float64),\n                    pl.col(pl.String),\n                    pl.col(pl.Boolean))\n           )","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:21.287062Z","iopub.execute_input":"2024-06-02T16:16:21.287475Z","iopub.status.idle":"2024-06-02T16:16:21.303217Z","shell.execute_reply.started":"2024-06-02T16:16:21.287436Z","shell.execute_reply":"2024-06-02T16:16:21.301889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_base.parquet'))\ntest = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_base.parquet'))\n\ntrain.write_parquet(os.path.join(SAVE_PATH, 'train/base/train_base.parquet'))\ntest.write_parquet(os.path.join(SAVE_PATH, 'test/base/test_base.parquet'))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:21.304719Z","iopub.execute_input":"2024-06-02T16:16:21.305155Z","iopub.status.idle":"2024-06-02T16:16:21.970314Z","shell.execute_reply.started":"2024-06-02T16:16:21.305117Z","shell.execute_reply":"2024-06-02T16:16:21.968767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg = ['actualdpd_943P','annuity_853A', 'byoccupationinc_3656910L', 'childnum_21L', 'credacc_actualbalance_314A',\n       'credacc_credlmt_575A','credacc_maxhisbal_375A','credacc_minhisbal_90A', 'credacc_transactions_402L', 'credamount_590A',\n       'currdebt_94A', 'mainoccupationinc_437A', 'pmtnum_8L','tenor_203L']\n\nmode = ['credtype_587L', 'district_544M', 'downpmt_134A', 'education_1138M', 'familystate_726L',\n        'inittransactioncode_279L', 'rejectreason_755M', 'status_219L']\n\ntrain_0 = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_applprev_1_0.parquet'))\nselect_columns = clean_columns(train_0, 0.1, train_0.columns)\ntrain_0 = train_0.select( pl.col(select_columns))\ntrain_0 = fill_na(train_0)\n\ntrain_0_ready = (train_0\n .group_by(\"case_id\")\n .agg(pl.col([col for col in avg if (col in select_columns)]).mean(),\n      pl.col([col for col in mode if col in select_columns]).mode().first())\n .sort(\"case_id\")\n )\n\ntrain_1 = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_applprev_1_1.parquet'))\ntrain_1 = train_1.select( pl.col(select_columns))\ntrain_1 = fill_na(train_1)\n\ntrain_1_ready = (train_1\n .group_by(\"case_id\")\n .agg(pl.col([col for col in avg if (col in select_columns)]).mean(),\n      pl.col([col for col in mode if col in select_columns]).mode().first())\n .sort(\"case_id\")\n )\n\ntrain = pl.concat([train_0_ready, train_1_ready]).group_by(\"case_id\").agg(pl.all().mode().first()) \ntrain.write_parquet(os.path.join(SAVE_PATH, \"train/applprev_1/train_applprev_1.parquet\"))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:16:21.972250Z","iopub.execute_input":"2024-06-02T16:16:21.972702Z","iopub.status.idle":"2024-06-02T16:17:05.557979Z","shell.execute_reply.started":"2024-06-02T16:16:21.972662Z","shell.execute_reply":"2024-06-02T16:17:05.556721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_applprev_1_0\ntest_0 = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_applprev_1_0.parquet'))\ntest_0 = test_0.select( pl.col(select_columns))\ntest_0 = fill_na(test_0)\n\ntest_0_ready = (test_0\n .group_by(\"case_id\")\n .agg(pl.col([col for col in avg if (col in select_columns)]).mean(),\n      pl.col([col for col in mode if col in select_columns]).mode().first())\n .sort(\"case_id\")\n#  .write_parquet(os.path.join(SAVE_PATH, 'test/applprev_1/test_applprev_1_0.parquet'))\n)\n\n#test_applprev_1_1\ntest_1 = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_applprev_1_1.parquet'))\ntest_1 = test_1.select( pl.col(select_columns))\ntest_1 = fill_na(test_1)\n\ntest_1_ready = (test_1\n .group_by(\"case_id\")\n .agg(pl.col([col for col in avg if (col in select_columns)]).mean(),\n      pl.col([col for col in mode if col in select_columns]).mode().first())\n .sort(\"case_id\")\n#  .write_parquet(os.path.join(SAVE_PATH, 'test/applprev_1/test_applprev_1_1.parquet'))\n )\n\n#test_applprev_1_2\ntest_2 = pl.read_parquet(os.path.join(DATA_PATH,'test/test_applprev_1_2.parquet'))\ntest_2 = test_2.select( pl.col(select_columns))\ntest_2 = fill_na(test_2)\n''\ntest_2_ready = (test_2\n .group_by(\"case_id\")\n .agg(pl.col([col for col in avg if (col in select_columns)]).mean(),\n      pl.col([col for col in mode if col in select_columns]).mode().first())\n .sort(\"case_id\")\n#  .write_parquet(os.path.join(SAVE_PATH, 'test/applprev_1/test_applprev_1_2.parquet'))\n )\n\ntest = pl.concat([test_0_ready, test_1_ready, test_2_ready]).group_by(\"case_id\").agg(pl.all().mode().first()) \ntrain.write_parquet(os.path.join(SAVE_PATH, \"test/applprev_1/test_applprev_1.parquet\"))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:05.563472Z","iopub.execute_input":"2024-06-02T16:17:05.563881Z","iopub.status.idle":"2024-06-02T16:17:06.742975Z","shell.execute_reply.started":"2024-06-02T16:17:05.563849Z","shell.execute_reply":"2024-06-02T16:17:06.741830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_tax_registry\ntrain_a = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_tax_registry_a_1.parquet'))\ntrain_b = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_tax_registry_b_1.parquet'))\ntrain_c = pl.read_parquet(os.path.join(DATA_PATH, 'train/train_tax_registry_c_1.parquet'))\n\ntrain_b = train_b.rename({\n    \"amount_4917619A\": \"amount_4527230A\",\n    \"name_4917606M\": \"name_4527232M\",\n    \"deductiondate_4917603D\": \"recorddate_4527225D\"\n})\ntrain_c = train_c.rename({\n    \"pmtamount_36A\": \"amount_4527230A\",\n    \"employername_160M\": \"name_4527232M\",\n    \"processingdate_168D\": \"recorddate_4527225D\"\n})\n\nselect_columns = clean_columns(train_a, 0.1, train_a.columns)\n\ntrain_a = train_a.select(pl.col(select_columns))\ntrain_b = train_b.select(pl.col(select_columns))\ntrain_c = train_c.select(pl.col(select_columns))\n\n#fill_na\ntrain_a = fill_na(train_a)\ntrain_b = fill_na(train_b)\ntrain_c = fill_na(train_c)\n\n#first file in train registry\ntrain_a = (train_a\n .group_by(\"case_id\")\n .agg(pl.col(pl.Float64, pl.Int64).exclude(\"case_id\", \"num_group1\").mean(),\n      pl.col(pl.String, pl.Boolean).mode().first())\n#  .select(pl.all().is_nan().sum())\n .select(pl.all().exclude(\"recorddate_4527225D\")) #this feature is not of much use in absence of some other date feature to compare it with\n .sort(\"case_id\")\n )\n\n#second file in train registry\ntrain_b = (train_b\n .group_by(\"case_id\")\n .agg(pl.col(pl.Float64, pl.Int64).exclude(\"case_id\", \"num_group1\").mean(),\n      pl.col(pl.String, pl.Boolean).mode().first())\n#  .select(pl.all().is_nan().sum())\n .select(pl.all().exclude(\"recorddate_4527225D\")) #this feature is not of much use in absence of some other date feature to compare it with\n .sort(\"case_id\")\n )\n\n#third file in train registry\ntrain_c = (train_c\n .group_by(\"case_id\")\n .agg(pl.col(pl.Float64, pl.Int64).exclude(\"case_id\", \"num_group1\").mean(),\n      pl.col(pl.String, pl.Boolean).mode().first())\n#  .select(pl.all().is_nan().sum())\n .select(pl.all().exclude(\"recorddate_4527225D\")) #this feature is not of much use in absence of some other date feature to compare it with\n .sort(\"case_id\")\n )\n\n#a, b and c contain some records having same case_id\ntrain = pl.concat([train_a, train_b, train_c]).group_by(\"case_id\").agg(pl.all().mode().first()) \ntrain.write_parquet(os.path.join(SAVE_PATH, 'train/tax_registry/train_tax_registry.parquet'))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:06.747161Z","iopub.execute_input":"2024-06-02T16:17:06.747514Z","iopub.status.idle":"2024-06-02T16:17:17.465277Z","shell.execute_reply.started":"2024-06-02T16:17:06.747483Z","shell.execute_reply":"2024-06-02T16:17:17.464206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test tax registry\ntest_a = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_tax_registry_a_1.parquet'))\ntest_b = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_tax_registry_b_1.parquet'))\ntest_c = pl.read_parquet(os.path.join(DATA_PATH, 'test/test_tax_registry_c_1.parquet')) #is empty\n\ntest_b = test_b.rename({\n    \"amount_4917619A\": \"amount_4527230A\",\n    \"name_4917606M\": \"name_4527232M\",\n    \"deductiondate_4917603D\": \"recorddate_4527225D\"\n})\n\n#fill_na\ntest_a = fill_na(test_a)\ntest_b = fill_na(test_b)\n\n#grouping and aggregating\ntest_a = (test_a\n .group_by(\"case_id\")\n .agg(pl.col(pl.Float64, pl.Int64).exclude(\"case_id\", \"num_group1\").mean(),\n      pl.col(pl.String, pl.Boolean).mode().first())\n#  .select(pl.all().is_nan().sum())\n .select(pl.all().exclude(\"recorddate_4527225D\")) #this feature is not of much use in absence of some other date feature to compare it with\n .sort(\"case_id\")\n )\n\ntest_b = (test_b\n .group_by(\"case_id\")\n .agg(pl.col(pl.Float64, pl.Int64).exclude(\"case_id\", \"num_group1\").mean(),\n      pl.col(pl.String, pl.Boolean).mode().first())\n#  .select(pl.all().is_nan().sum())\n .select(pl.all().exclude(\"recorddate_4527225D\")) #this feature is not of much use in absence of some other date feature to compare it with\n .sort(\"case_id\")\n )\n\n#writing to disk\ntest = pl.concat([test_a, test_b]).group_by(\"case_id\").agg(pl.all().mode().first()) #a and b contain some records having same case_id\ntest.write_parquet(os.path.join(SAVE_PATH, 'test/tax_registry/test_tax_registry.parquet'))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:17.466788Z","iopub.execute_input":"2024-06-02T16:17:17.467246Z","iopub.status.idle":"2024-06-02T16:17:17.502894Z","shell.execute_reply.started":"2024-06-02T16:17:17.467202Z","shell.execute_reply":"2024-06-02T16:17:17.501747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = pl.read_parquet(os.path.join(SAVE_PATH, \"train/base/train_base.parquet\"))\ntax_registry = pl.read_parquet(os.path.join(SAVE_PATH, \"train/tax_registry/train_tax_registry.parquet\"))\n# applprev_1 = pl.read_parquet(os.path.join(SAVE_PATH, \"train/applprev_1/train_applprev_1.parquet\"))\n\ntrain = (base\n         .join(tax_registry, on=\"case_id\", how=\"left\")\n#          .join(applprev_1, on=\"case_id\", how=\"inner\")\n         .select(pl.all().exclude(\"date_decision\"))\n         )\ntrain = fill_na(train)\n\ntrain_cat = (train\n            .select(pl.col(\"case_id\"), pl.col(pl.String))\n            )\ntrain_num = (train\n             .select(pl.all().exclude(pl.String, pl.Boolean), pl.col(pl.Boolean).cast(pl.Int64))\n            )","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:17.504446Z","iopub.execute_input":"2024-06-02T16:17:17.504874Z","iopub.status.idle":"2024-06-02T16:17:18.424089Z","shell.execute_reply.started":"2024-06-02T16:17:17.504833Z","shell.execute_reply":"2024-06-02T16:17:18.422720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding and scaling\n\nfrom sklearn.preprocessing import OrdinalEncoder\nord_encoder = OrdinalEncoder()\n\ntrain_cat_encoded = pd.DataFrame(ord_encoder.fit_transform(train_cat.drop(\"case_id\")), columns=train_cat.columns[1:])\ntrain_cat_encoded[\"case_id\"] = train_cat[\"case_id\"]\n\ntrain_num_pd = train_num.to_pandas()\n\ntrain_encoded = (train_num_pd\n                 .merge(train_cat_encoded, how=\"inner\", on=\"case_id\")\n                )\n\nfrom sklearn.preprocessing import StandardScaler\nstd_scaler = StandardScaler()\nfeat_scale = [col for col in train_encoded.columns if col not in [\"case_id\", \"target\"]]\ntrain_encoded_scaled = pd.DataFrame(std_scaler.fit_transform(train_encoded[feat_scale]), columns=feat_scale)\ntrain_encoded_scaled[[\"case_id\", \"target\"]] = train_encoded[[\"case_id\", \"target\"]]","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:18.425648Z","iopub.execute_input":"2024-06-02T16:17:18.426027Z","iopub.status.idle":"2024-06-02T16:17:22.253777Z","shell.execute_reply.started":"2024-06-02T16:17:18.425982Z","shell.execute_reply":"2024-06-02T16:17:22.252687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_encoded_scaled","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:22.255467Z","iopub.execute_input":"2024-06-02T16:17:22.255819Z","iopub.status.idle":"2024-06-02T16:17:22.279507Z","shell.execute_reply.started":"2024-06-02T16:17:22.255789Z","shell.execute_reply":"2024-06-02T16:17:22.278339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #train test split\n\n# from sklearn.model_selection import train_test_split\n# X_train, X_val, y_train, y_val = train_test_split(train_encoded_scaled.drop(columns=[\"target\"]), \\\n#                                                   train_encoded_scaled[\"case_id\"], test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:22.280963Z","iopub.execute_input":"2024-06-02T16:17:22.281340Z","iopub.status.idle":"2024-06-02T16:17:22.286525Z","shell.execute_reply.started":"2024-06-02T16:17:22.281311Z","shell.execute_reply":"2024-06-02T16:17:22.285412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport vaex as vx\nimport vaex.ml\n","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:22.287843Z","iopub.execute_input":"2024-06-02T16:17:22.288181Z","iopub.status.idle":"2024-06-02T16:17:23.892372Z","shell.execute_reply.started":"2024-06-02T16:17:22.288154Z","shell.execute_reply":"2024-06-02T16:17:23.890981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model training\n\nfrom sklearn.linear_model import LogisticRegression\nlog_reg = LogisticRegression(max_iter=100)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:23.894286Z","iopub.execute_input":"2024-06-02T16:17:23.895616Z","iopub.status.idle":"2024-06-02T16:17:24.127457Z","shell.execute_reply.started":"2024-06-02T16:17:23.895564Z","shell.execute_reply":"2024-06-02T16:17:24.126222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = vx.from_pandas(train_encoded_scaled)\nfrom vaex.ml.sklearn import IncrementalPredictor\nfrom sklearn.linear_model import SGDClassifier\nfeatures = [col for col in train.columns if col not in [\"case_id\", \"target\"]]\ntarget = 'target'\nmodel = SGDClassifier(loss=\"modified_huber\")\nvaex_model = IncrementalPredictor(model=model, features=features, target=target, batch_size=100000,\n                                 partial_fit_kwargs={'classes':[0, 1]})\n","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:24.128968Z","iopub.execute_input":"2024-06-02T16:17:24.129405Z","iopub.status.idle":"2024-06-02T16:17:24.173419Z","shell.execute_reply.started":"2024-06-02T16:17:24.129369Z","shell.execute_reply":"2024-06-02T16:17:24.172073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vaex_model.fit(df=train, progress='widget')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:24.174974Z","iopub.execute_input":"2024-06-02T16:17:24.175403Z","iopub.status.idle":"2024-06-02T16:17:25.007156Z","shell.execute_reply.started":"2024-06-02T16:17:24.175366Z","shell.execute_reply":"2024-06-02T16:17:25.006050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"\nbase = pl.read_parquet(os.path.join(SAVE_PATH, \"test/base/test_base.parquet\"))\ntax_registry = pl.read_parquet(os.path.join(SAVE_PATH, \"test/tax_registry/test_tax_registry.parquet\"))\n# applprev_1 = pl.read_parquet(os.path.join(SAVE_PATH, \"test/applprev_1/test_applprev_1.parquet\"))\n\n\ntest = (base\n         .join(tax_registry, on=\"case_id\", how=\"left\")\n#          .join(applprev_1, on=\"case_id\", how=\"inner\")\n         .select(pl.all().exclude(\"date_decision\"))\n         )\n\ntest = fill_na(test)\n\ntest_cat = (test\n                .select(pl.col(\"case_id\"), pl.col(pl.String))\n                )\n\ntest_num = (test\n             .select(pl.all().exclude(pl.String, pl.Boolean), pl.col(pl.Boolean).cast(pl.Int64))\n            )","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:17:25.008474Z","iopub.execute_input":"2024-06-02T16:17:25.008834Z","iopub.status.idle":"2024-06-02T16:17:25.033207Z","shell.execute_reply.started":"2024-06-02T16:17:25.008803Z","shell.execute_reply":"2024-06-02T16:17:25.031965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the error is in one of these two lines\n\n# try:\ntemp_df = test_cat.drop(\"case_id\").to_pandas()\ntest_cat_np = ord_encoder.fit_transform(temp_df)\ntest_cat_encoded = pd.DataFrame(test_cat_np, columns=test_cat.columns[1:])\ntest_cat_encoded[\"case_id\"] = test_cat[\"case_id\"]\ntest_num_pd = test_num.to_pandas()\n\ntest_encoded = (test_num_pd\n                 .merge(test_cat_encoded, how=\"inner\", on=\"case_id\")\n                )\n\nfeat_scale = [col for col in test_encoded.columns if col not in [\"case_id\"]]\ntest_encoded_scaled = pd.DataFrame(std_scaler.transform(test_encoded[feat_scale]), columns=feat_scale)\ntest_encoded_scaled[[\"case_id\"]] = test_encoded[[\"case_id\"]]\n\ntest = vx.from_pandas(test_encoded_scaled)\n\nsubmission = pd.DataFrame({\n    \"case_id\": test_encoded_scaled[\"case_id\"],\n    \"score\": vaex_model.predict(test)\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)\n# except:\n#     print(\"No\")","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:41:28.160715Z","iopub.execute_input":"2024-06-02T16:41:28.161271Z","iopub.status.idle":"2024-06-02T16:41:28.233689Z","shell.execute_reply.started":"2024-06-02T16:41:28.161229Z","shell.execute_reply":"2024-06-02T16:41:28.232583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}