{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt \nimport plotly.express as px  # Add this import statement\nimport seaborn as sns \n\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-11T23:19:12.015659Z","iopub.execute_input":"2024-02-11T23:19:12.016102Z","iopub.status.idle":"2024-02-11T23:19:12.022391Z","shell.execute_reply.started":"2024-02-11T23:19:12.016042Z","shell.execute_reply":"2024-02-11T23:19:12.021219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 1: Data Exploration and Understanding\n ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns  \n\ntrain_base = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ntest_base = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:13.583374Z","iopub.execute_input":"2024-02-11T23:19:13.583759Z","iopub.status.idle":"2024-02-11T23:19:14.515724Z","shell.execute_reply.started":"2024-02-11T23:19:13.583728Z","shell.execute_reply":"2024-02-11T23:19:14.514810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:15.990934Z","iopub.execute_input":"2024-02-11T23:19:15.991360Z","iopub.status.idle":"2024-02-11T23:19:16.161243Z","shell.execute_reply.started":"2024-02-11T23:19:15.991319Z","shell.execute_reply":"2024-02-11T23:19:16.160089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_base.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:16.890968Z","iopub.execute_input":"2024-02-11T23:19:16.891701Z","iopub.status.idle":"2024-02-11T23:19:16.902994Z","shell.execute_reply.started":"2024-02-11T23:19:16.891668Z","shell.execute_reply":"2024-02-11T23:19:16.901698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:20.405604Z","iopub.execute_input":"2024-02-11T23:19:20.406431Z","iopub.status.idle":"2024-02-11T23:19:20.984811Z","shell.execute_reply.started":"2024-02-11T23:19:20.406394Z","shell.execute_reply":"2024-02-11T23:19:20.984022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_base.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:20.986419Z","iopub.execute_input":"2024-02-11T23:19:20.986856Z","iopub.status.idle":"2024-02-11T23:19:21.012077Z","shell.execute_reply.started":"2024-02-11T23:19:20.986816Z","shell.execute_reply":"2024-02-11T23:19:21.010950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Convert 'date_decision' to datetime\ntrain_base['date_decision'] = pd.to_datetime(train_base['date_decision'])\ntest_base['date_decision'] = pd.to_datetime(test_base['date_decision'])\n\n# Check for missing values\nprint(\"Train Base Missing Values:\\n\", train_base.isnull().sum())\nprint(\"\\nTest Base Missing Values:\\n\", test_base.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:21:27.219484Z","iopub.execute_input":"2024-02-11T23:21:27.219915Z","iopub.status.idle":"2024-02-11T23:21:27.531420Z","shell.execute_reply.started":"2024-02-11T23:21:27.219882Z","shell.execute_reply":"2024-02-11T23:21:27.530443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2: Feature Engineering\n","metadata":{}},{"cell_type":"code","source":"# Convert 'date_decision' column to datetime format\ntrain_base['date_decision'] = pd.to_datetime(train_base['date_decision'])\ntest_base['date_decision'] = pd.to_datetime(test_base['date_decision'])\n\n# Feature engineering for time-based variables\ntrain_base['day_of_week'] = train_base['date_decision'].dt.dayofweek\ntrain_base['day_of_month'] = train_base['date_decision'].dt.day\ntrain_base['quarter'] = train_base['date_decision'].dt.quarter\ntrain_base['year'] = train_base['date_decision'].dt.year\n\ntest_base['day_of_week'] = test_base['date_decision'].dt.dayofweek\ntest_base['day_of_month'] = test_base['date_decision'].dt.day\ntest_base['quarter'] = test_base['date_decision'].dt.quarter\ntest_base['year'] = test_base['date_decision'].dt.year\n\n# Display the first few rows of the datasets after feature engineering\nprint(\"\\nFeature-Engineered Train Base:\\n\", train_base.head())\nprint(\"\\nFeature-Engineered Test Base:\\n\", test_base.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:23:44.566786Z","iopub.execute_input":"2024-02-11T23:23:44.567521Z","iopub.status.idle":"2024-02-11T23:23:44.818720Z","shell.execute_reply.started":"2024-02-11T23:23:44.567468Z","shell.execute_reply":"2024-02-11T23:23:44.817639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 3: virtualization\n","metadata":{}},{"cell_type":"code","source":"# # Function to create individual plots for each column\n# def plot_columns(data, title_prefix):\n#     for column in data.columns:\n#         fig = px.histogram(data, x=column, title=f\"{title_prefix} - {column}\")\n#         fig.show()\n\n# # Create individual plots for train_base\n# plot_columns(train_base, 'Train Base')\n\n# # Create individual plots for test_base\n# # plot_columns(test_base, 'Test Base')\n","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:19:51.664230Z","iopub.execute_input":"2024-02-11T23:19:51.664626Z","iopub.status.idle":"2024-02-11T23:19:51.669789Z","shell.execute_reply.started":"2024-02-11T23:19:51.664596Z","shell.execute_reply":"2024-02-11T23:19:51.668647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"case_id\tdate_decision\tMONTH\tWEEK_NUM\ttargetcase_id\tdate_decision\tMONTH\tWEEK_NUM\ttarget# Step 4: modeling XGB \n","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\n# Define the features (X) and target variable (y) for training\nX_train = train_base.drop(['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target'], axis=1)\ny_train = train_base['target']\n\n# Initialize and train the XGBoost model\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:23:50.292011Z","iopub.execute_input":"2024-02-11T23:23:50.292396Z","iopub.status.idle":"2024-02-11T23:23:55.112735Z","shell.execute_reply.started":"2024-02-11T23:23:50.292367Z","shell.execute_reply":"2024-02-11T23:23:55.111785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 5: submission\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Assuming you have the test set loaded into a DataFrame called test_base\n# Define the features (X) for testing\nX_test = test_base.drop(['case_id', 'date_decision', 'MONTH', 'WEEK_NUM'], axis=1)\n\n# Generate random scores for each record in the test set\ntest_pred = np.random.uniform(0, 1, len(X_test))\n\n# Assuming you have a 'case_id' column in the test_base DataFrame, create a submission DataFrame\nsubmission = pd.DataFrame({'case_id': test_base['case_id'], 'score': test_pred})\n\n# Save the submission to a CSV file\nsubmission.to_csv('submission.csv', index=None)\nsubmission.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-11T23:25:13.182169Z","iopub.execute_input":"2024-02-11T23:25:13.182605Z","iopub.status.idle":"2024-02-11T23:25:13.199498Z","shell.execute_reply.started":"2024-02-11T23:25:13.182574Z","shell.execute_reply":"2024-02-11T23:25:13.198162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}