{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-20T15:01:29.499682Z","iopub.execute_input":"2024-03-20T15:01:29.500445Z","iopub.status.idle":"2024-03-20T15:01:29.517153Z","shell.execute_reply.started":"2024-03-20T15:01:29.500402Z","shell.execute_reply":"2024-03-20T15:01:29.515222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setting up","metadata":{}},{"cell_type":"code","source":"df1 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv')\ndf2 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_applprev_1_0.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:01:29.520669Z","iopub.execute_input":"2024-03-20T15:01:29.521676Z","iopub.status.idle":"2024-03-20T15:02:13.898274Z","shell.execute_reply.started":"2024-03-20T15:01:29.521615Z","shell.execute_reply":"2024-03-20T15:02:13.896520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.merge(df1, df2, on=\"case_id\", how=\"left\").sample(10000)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:13.900148Z","iopub.execute_input":"2024-03-20T15:02:13.901105Z","iopub.status.idle":"2024-03-20T15:02:23.033655Z","shell.execute_reply.started":"2024-03-20T15:02:13.901051Z","shell.execute_reply":"2024-03-20T15:02:23.032236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:23.036950Z","iopub.execute_input":"2024-03-20T15:02:23.037504Z","iopub.status.idle":"2024-03-20T15:02:23.089565Z","shell.execute_reply.started":"2024-03-20T15:02:23.037455Z","shell.execute_reply":"2024-03-20T15:02:23.088228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Features","metadata":{}},{"cell_type":"code","source":"from sklearn.compose import make_column_selector as selector\n\ncategorical_columns_selector = selector(dtype_include=object)\ncategorical_columns = categorical_columns_selector(df)\ncategorical_columns","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:23.091278Z","iopub.execute_input":"2024-03-20T15:02:23.091673Z","iopub.status.idle":"2024-03-20T15:02:23.667392Z","shell.execute_reply.started":"2024-03-20T15:02:23.091640Z","shell.execute_reply":"2024-03-20T15:02:23.666375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_encoded = pd.get_dummies(df, columns=categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:23.668947Z","iopub.execute_input":"2024-03-20T15:02:23.669478Z","iopub.status.idle":"2024-03-20T15:02:24.288976Z","shell.execute_reply.started":"2024-03-20T15:02:23.669442Z","shell.execute_reply":"2024-03-20T15:02:24.287311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_encoded.fillna(df_encoded.mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:24.291406Z","iopub.execute_input":"2024-03-20T15:02:24.293218Z","iopub.status.idle":"2024-03-20T15:02:29.279546Z","shell.execute_reply.started":"2024-03-20T15:02:24.293141Z","shell.execute_reply":"2024-03-20T15:02:29.277541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_encoded","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:29.281591Z","iopub.execute_input":"2024-03-20T15:02:29.282014Z","iopub.status.idle":"2024-03-20T15:02:29.323413Z","shell.execute_reply.started":"2024-03-20T15:02:29.281981Z","shell.execute_reply":"2024-03-20T15:02:29.322180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model hyperparameters","metadata":{}},{"cell_type":"code","source":"import sklearn.datasets\nimport sklearn.ensemble\nimport sklearn.model_selection\n\n\ndef objective():\n    dataset = df_encoded  # Prepare the data.\n\n    rgr = sklearn.linear_model.LogisticRegression()  # Define the model.\n\n    return sklearn.model_selection.cross_val_score(\n        rgr, dataset.drop(columns=['target']), dataset['target'], n_jobs=-1, cv=3\n    ).mean()  # Train and evaluate the model.\n\n\nprint(\"Accuracy: {}\".format(objective()))","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:29.327377Z","iopub.execute_input":"2024-03-20T15:02:29.327926Z","iopub.status.idle":"2024-03-20T15:02:54.912301Z","shell.execute_reply.started":"2024-03-20T15:02:29.327879Z","shell.execute_reply":"2024-03-20T15:02:54.911200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\n\n\ndef objective(trial):\n    dataset = df_encoded\n\n    tol = trial.suggest_float(\"tol\", 1e-5, 1e-3)\n    C = trial.suggest_float(\"C\", 0.1, 5)\n\n    rgr = sklearn.linear_model.LogisticRegression(tol=tol, C=C)\n\n    return sklearn.model_selection.cross_val_score(\n        rgr, dataset.drop(columns=['target']), dataset['target'], n_jobs=-1, cv=3\n    ).mean()\n\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=10)\n\ntrial = study.best_trial\n\nprint(\"Accuracy: {}\".format(trial.value))\nprint(\"Best hyperparameters: {}\".format(trial.params))","metadata":{"execution":{"iopub.status.busy":"2024-03-20T15:02:54.913969Z","iopub.execute_input":"2024-03-20T15:02:54.914851Z","iopub.status.idle":"2024-03-20T15:06:28.697056Z","shell.execute_reply.started":"2024-03-20T15:02:54.914812Z","shell.execute_reply":"2024-03-20T15:06:28.696248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}