{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Example Notebook**\nLet me tell you our idea, means what we are gonna to do in our Machine Learning Model of Home Credit 2024, in which we have to predict whether a userl will default the loan or not! So, the content of this notebook is listed below -:\n\n* Import The Data\n* Do Train Test Split\n* Do Feature Engineering\n* Train the model from Random Forest Regressor\n* Test it! **Complete!**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport dask.dataframe as dd\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\n\nfrom sklearn.ensemble import RandomForestRegressor\n\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.base import TransformerMixin","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-10T08:47:32.690400Z","iopub.execute_input":"2024-02-10T08:47:32.690854Z","iopub.status.idle":"2024-02-10T08:47:32.700066Z","shell.execute_reply.started":"2024-02-10T08:47:32.690820Z","shell.execute_reply":"2024-02-10T08:47:32.698274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_base1 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv', nrows=10000, low_memory=False)\ndf_static_0 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_0.csv', nrows=10000, low_memory=False)\ndf_static_1 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_1.csv', nrows=10000, low_memory=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:06.272957Z","iopub.execute_input":"2024-02-10T09:55:06.276405Z","iopub.status.idle":"2024-02-10T09:55:06.811753Z","shell.execute_reply.started":"2024-02-10T09:55:06.276345Z","shell.execute_reply":"2024-02-10T09:55:06.810455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_static1 = pd.concat([df_static_0,df_static_1])\ndel(df_static_0,df_static_1)\n\ntrain_df = pd.merge(df_train_base1, df_train_static1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:08.372803Z","iopub.execute_input":"2024-02-10T09:55:08.373246Z","iopub.status.idle":"2024-02-10T09:55:08.484376Z","shell.execute_reply.started":"2024-02-10T09:55:08.373210Z","shell.execute_reply":"2024-02-10T09:55:08.482667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_base = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv')\ndf_test_static_0 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_0.csv')\ndf_test_static_1 = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_1.csv')\n\ndf_test_static = pd.concat([df_test_static_0, df_test_static_1])\ndel(df_test_static_0, df_test_static_1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:32.934171Z","iopub.execute_input":"2024-02-10T09:55:32.934668Z","iopub.status.idle":"2024-02-10T09:55:32.974445Z","shell.execute_reply.started":"2024-02-10T09:55:32.934632Z","shell.execute_reply":"2024-02-10T09:55:32.972970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.merge(df_test_base, df_test_static)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:34.480832Z","iopub.execute_input":"2024-02-10T09:55:34.481342Z","iopub.status.idle":"2024-02-10T09:55:34.495808Z","shell.execute_reply.started":"2024-02-10T09:55:34.481304Z","shell.execute_reply":"2024-02-10T09:55:34.494447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:40.274815Z","iopub.execute_input":"2024-02-10T09:55:40.276186Z","iopub.status.idle":"2024-02-10T09:55:40.305363Z","shell.execute_reply.started":"2024-02-10T09:55:40.276114Z","shell.execute_reply":"2024-02-10T09:55:40.304417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Date Decision'] = pd.to_datetime(train_df['date_decision'])\ntrain_df['Month'] = train_df['Date Decision'].dt.month\ntrain_df = train_df.drop(columns=['date_decision', 'MONTH'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:45.661511Z","iopub.execute_input":"2024-02-10T09:55:45.661915Z","iopub.status.idle":"2024-02-10T09:55:45.682163Z","shell.execute_reply.started":"2024-02-10T09:55:45.661886Z","shell.execute_reply":"2024-02-10T09:55:45.680724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Date Decision'] = pd.to_datetime(test_data['date_decision'])\ntest_data['Month'] = test_data['Date Decision'].dt.month\ntest_data = test_data.drop(columns=['date_decision', 'MONTH'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:55:50.262262Z","iopub.execute_input":"2024-02-10T09:55:50.262667Z","iopub.status.idle":"2024-02-10T09:55:50.274770Z","shell.execute_reply.started":"2024-02-10T09:55:50.262637Z","shell.execute_reply":"2024-02-10T09:55:50.273514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns=['Date Decision'], axis=1)\ntest_data = test_data.drop(columns=['Date Decision'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:18.118228Z","iopub.execute_input":"2024-02-10T09:56:18.119370Z","iopub.status.idle":"2024-02-10T09:56:18.138585Z","shell.execute_reply.started":"2024-02-10T09:56:18.119330Z","shell.execute_reply":"2024-02-10T09:56:18.136660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, test_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:19.519665Z","iopub.execute_input":"2024-02-10T09:56:19.520067Z","iopub.status.idle":"2024-02-10T09:56:19.532933Z","shell.execute_reply.started":"2024-02-10T09:56:19.520037Z","shell.execute_reply":"2024-02-10T09:56:19.531345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['target'], axis=1)\ny = train_df['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:24.510510Z","iopub.execute_input":"2024-02-10T09:56:24.510914Z","iopub.status.idle":"2024-02-10T09:56:24.541885Z","shell.execute_reply.started":"2024-02-10T09:56:24.510884Z","shell.execute_reply":"2024-02-10T09:56:24.540963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:25.918606Z","iopub.execute_input":"2024-02-10T09:56:25.919063Z","iopub.status.idle":"2024-02-10T09:56:25.952352Z","shell.execute_reply.started":"2024-02-10T09:56:25.919029Z","shell.execute_reply":"2024-02-10T09:56:25.951040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:34.427252Z","iopub.execute_input":"2024-02-10T09:56:34.427750Z","iopub.status.idle":"2024-02-10T09:56:34.436000Z","shell.execute_reply.started":"2024-02-10T09:56:34.427706Z","shell.execute_reply":"2024-02-10T09:56:34.434490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:36.444573Z","iopub.execute_input":"2024-02-10T09:56:36.445016Z","iopub.status.idle":"2024-02-10T09:56:36.453620Z","shell.execute_reply.started":"2024-02-10T09:56:36.444976Z","shell.execute_reply":"2024-02-10T09:56:36.452408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:38.103721Z","iopub.execute_input":"2024-02-10T09:56:38.104844Z","iopub.status.idle":"2024-02-10T09:56:38.115502Z","shell.execute_reply.started":"2024-02-10T09:56:38.104800Z","shell.execute_reply":"2024-02-10T09:56:38.113850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_columns_X_train = X_train.select_dtypes(include='number').columns.tolist()\nnumeric_columns_X_test = X_test.select_dtypes(include='number').columns.tolist()\n\ncategorical_columns_X_train = X_train.select_dtypes(exclude='number').columns.tolist()\ncategorical_columns_X_test = X_test.select_dtypes(exclude='number').columns.tolist()\n\nprint(len(numeric_columns_X_train))\nprint(len(categorical_columns_X_train))","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:39.740086Z","iopub.execute_input":"2024-02-10T09:56:39.740547Z","iopub.status.idle":"2024-02-10T09:56:39.780011Z","shell.execute_reply.started":"2024-02-10T09:56:39.740505Z","shell.execute_reply":"2024-02-10T09:56:39.778660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns_finite = [col for col in categorical_columns_X_train if X_train[col].nunique() <= 6]\nselected_categorical_columns = [col for col in categorical_columns_X_train if X_train[col].nunique() > 10]\ntop_values_count = 5\nselected_categorical_columns_to_encode = [col for col in selected_categorical_columns if X_train[col].nunique() >= top_values_count]","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:41.695663Z","iopub.execute_input":"2024-02-10T09:56:41.696764Z","iopub.status.idle":"2024-02-10T09:56:41.762029Z","shell.execute_reply.started":"2024-02-10T09:56:41.696712Z","shell.execute_reply":"2024-02-10T09:56:41.761066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_columns = numeric_columns_X_train\n\nnumerical_columns_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\ncategorical_columns_finite = categorical_columns_finite\n\ncategorical_columns_finite_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('ohe_finite', OneHotEncoder(handle_unknown='ignore'))\n])\n\ncategorical_columns_infinite = selected_categorical_columns_to_encode\n\ncategorical_columns_infinite_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('ohe_infinite', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:43.370517Z","iopub.execute_input":"2024-02-10T09:56:43.374396Z","iopub.status.idle":"2024-02-10T09:56:43.383519Z","shell.execute_reply.started":"2024-02-10T09:56:43.374338Z","shell.execute_reply":"2024-02-10T09:56:43.382347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_regressor = RandomForestRegressor(n_estimators=100, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:45.531420Z","iopub.execute_input":"2024-02-10T09:56:45.532155Z","iopub.status.idle":"2024-02-10T09:56:45.537491Z","shell.execute_reply.started":"2024-02-10T09:56:45.532093Z","shell.execute_reply":"2024-02-10T09:56:45.535968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers = [\n        ('Numerical Transformer', numerical_columns_transformer, numerical_columns),\n        ('OneHotEncoding Finite', categorical_columns_finite_transformer, categorical_columns_finite),\n        ('OneHotEncoding Infinite', categorical_columns_infinite_transformer, categorical_columns_infinite)\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:47.828748Z","iopub.execute_input":"2024-02-10T09:56:47.829215Z","iopub.status.idle":"2024-02-10T09:56:47.836515Z","shell.execute_reply.started":"2024-02-10T09:56:47.829179Z","shell.execute_reply":"2024-02-10T09:56:47.834220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('RandomForestRegressor', rf_regressor)\n])","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:49.495952Z","iopub.execute_input":"2024-02-10T09:56:49.496409Z","iopub.status.idle":"2024-02-10T09:56:49.502050Z","shell.execute_reply.started":"2024-02-10T09:56:49.496359Z","shell.execute_reply":"2024-02-10T09:56:49.500773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T09:56:51.931901Z","iopub.execute_input":"2024-02-10T09:56:51.932357Z","iopub.status.idle":"2024-02-10T09:59:04.993520Z","shell.execute_reply.started":"2024-02-10T09:56:51.932322Z","shell.execute_reply":"2024-02-10T09:59:04.992146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_for_Test_data = pipe.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T10:04:36.952993Z","iopub.execute_input":"2024-02-10T10:04:36.953572Z","iopub.status.idle":"2024-02-10T10:04:36.996365Z","shell.execute_reply.started":"2024-02-10T10:04:36.953534Z","shell.execute_reply":"2024-02-10T10:04:36.994990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\n    'case_id': test_data['case_id'],\n    'target': predictions_for_Test_data\n})\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T10:04:50.285400Z","iopub.execute_input":"2024-02-10T10:04:50.285863Z","iopub.status.idle":"2024-02-10T10:04:50.295427Z","shell.execute_reply.started":"2024-02-10T10:04:50.285826Z","shell.execute_reply":"2024-02-10T10:04:50.293944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}