{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-06T20:26:33.498893Z","iopub.execute_input":"2024-04-06T20:26:33.500076Z","iopub.status.idle":"2024-04-06T20:26:34.760022Z","shell.execute_reply.started":"2024-04-06T20:26:33.500034Z","shell.execute_reply":"2024-04-06T20:26:34.758737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Read the CSV file into a DataFrame\ndf = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv')\n\n# Display the first few rows of the DataFrame\nprint(df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:31:55.39062Z","iopub.execute_input":"2024-04-06T20:31:55.391655Z","iopub.status.idle":"2024-04-06T20:31:56.901425Z","shell.execute_reply.started":"2024-04-06T20:31:55.391614Z","shell.execute_reply":"2024-04-06T20:31:56.900166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-04-06T11:43:14.352192Z","iopub.execute_input":"2024-04-06T11:43:14.353572Z","iopub.status.idle":"2024-04-06T11:43:14.37272Z","shell.execute_reply.started":"2024-04-06T11:43:14.353524Z","shell.execute_reply":"2024-04-06T11:43:14.3718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_counts = df['target'].value_counts()\n\n# Display the counts of 0s and 1s in the \"target\" column\nprint(\"Count of 0s:\", target_counts[0])\nprint(\"Count of 1s:\", target_counts[1])","metadata":{"execution":{"iopub.status.busy":"2024-04-06T11:44:37.864514Z","iopub.execute_input":"2024-04-06T11:44:37.86567Z","iopub.status.idle":"2024-04-06T11:44:37.897075Z","shell.execute_reply.started":"2024-04-06T11:44:37.865618Z","shell.execute_reply":"2024-04-06T11:44:37.895689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport dask.dataframe as dd\n\n# Load the two dataframes\nTrain_applprev_1_0 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_applprev_1_0.csv\")\nTrain_applprev_1_1 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_applprev_1_1.csv\")\n\n# Concatenate the two dataframes into one\ncombined_df = dd.concat([Train_applprev_1_0, Train_applprev_1_1])\n\n# Select only the 'case_id' and 'actualdpd_943P' columns\nselected_columns_df = combined_df[['case_id', 'actualdpd_943P']]\n\n# Drop the rest of the columns\nselected_columns_df = selected_columns_df.compute()  # This will compute the result in parallel\n\n# Print the resulting dataframe\nprint(selected_columns_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T12:05:20.735575Z","iopub.execute_input":"2024-04-06T12:05:20.736048Z","iopub.status.idle":"2024-04-06T12:05:31.292045Z","shell.execute_reply.started":"2024-04-06T12:05:20.736012Z","shell.execute_reply":"2024-04-06T12:05:31.29051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_columns_df","metadata":{"execution":{"iopub.status.busy":"2024-04-06T12:11:27.74242Z","iopub.execute_input":"2024-04-06T12:11:27.743035Z","iopub.status.idle":"2024-04-06T12:11:27.761145Z","shell.execute_reply.started":"2024-04-06T12:11:27.742995Z","shell.execute_reply":"2024-04-06T12:11:27.759521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\n# Create a scatter plot\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x=selected_columns_df['case_id'], y=selected_columns_df['actualdpd_943P'], color='orange')\nplt.title('Scatter Plot of case_id vs actualdpd_943P')\nplt.xlabel('case_id')\nplt.ylabel('actualdpd_943P')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T12:21:51.022168Z","iopub.execute_input":"2024-04-06T12:21:51.022679Z","iopub.status.idle":"2024-04-06T12:22:06.419213Z","shell.execute_reply.started":"2024-04-06T12:21:51.022645Z","shell.execute_reply":"2024-04-06T12:22:06.417765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the total count of non-zero values in the 'actualdpd_943P' column\nnon_zero_count = (selected_columns_df['actualdpd_943P'] != 0).sum()\n\n# Print the total count of non-zero values\nprint(\"Total count of non-zero values in 'actualdpd_943P' column:\", non_zero_count)","metadata":{"execution":{"iopub.status.busy":"2024-04-06T12:28:58.934479Z","iopub.execute_input":"2024-04-06T12:28:58.934952Z","iopub.status.idle":"2024-04-06T12:28:58.955778Z","shell.execute_reply.started":"2024-04-06T12:28:58.934919Z","shell.execute_reply":"2024-04-06T12:28:58.954433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\n\n# Load the first CSV file\nddf1 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_0.csv\")\n\n# Load the second CSV file\nddf2 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_1.csv\")\n\n# Concatenate both dataframes\ncombined_ddf = dd.concat([ddf1, ddf2])\n\n# Keep only the columns 'case_id' and 'actualdpdtolerance_344P'\ncombined_ddf = combined_ddf[['case_id', 'actualdpdtolerance_344P']]\n\n# Convert back to Pandas dataframe if needed\ncombined_df = combined_ddf.compute()\n\n# Show the resulting dataframe\nprint(combined_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T12:34:57.715411Z","iopub.execute_input":"2024-04-06T12:34:57.715828Z","iopub.status.idle":"2024-04-06T12:35:08.246075Z","shell.execute_reply.started":"2024-04-06T12:34:57.7158Z","shell.execute_reply":"2024-04-06T12:35:08.245117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\n\n# Load the first CSV file\ndf1 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_0.csv\")\n\n# Load the second CSV file\ndf2 = dd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_1.csv\")\n\n# Concatenate the two DataFrames\ndpd_tolerance_df = dd.concat([df1, df2])\n\n# Select the main columns\ndpd_tolerance_df = dpd_tolerance_df[['actualdpdtolerance_344P', 'case_id']]\n\n# Convert to Pandas DataFrame\ndpd_tolerance_df = dpd_tolerance_df.compute()\n\n# Display the first few rows of the DataFrame\nprint(dpd_tolerance_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:32:10.309582Z","iopub.execute_input":"2024-04-06T20:32:10.309983Z","iopub.status.idle":"2024-04-06T20:32:12.268119Z","shell.execute_reply.started":"2024-04-06T20:32:10.309955Z","shell.execute_reply":"2024-04-06T20:32:12.266735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dpd_tolerance_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:34:15.990922Z","iopub.execute_input":"2024-04-06T20:34:15.991354Z","iopub.status.idle":"2024-04-06T20:34:15.999932Z","shell.execute_reply.started":"2024-04-06T20:34:15.99132Z","shell.execute_reply":"2024-04-06T20:34:15.998303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count of zero values\nzero_count = (dpd_tolerance_df == 0).sum()\n\n# Count of non-null values\nnon_null_count = dpd_tolerance_df.notnull().sum()\n\nprint(\"Count of Zero Values:\")\nprint(zero_count)\n\nprint(\"\\nCount of Non-Null Values:\")\nprint(non_null_count)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:33:05.637483Z","iopub.execute_input":"2024-04-06T20:33:05.637908Z","iopub.status.idle":"2024-04-06T20:33:05.649393Z","shell.execute_reply.started":"2024-04-06T20:33:05.637874Z","shell.execute_reply":"2024-04-06T20:33:05.648044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Read the CSV file into a DataFrame\ncredit_bureau_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_b_1.csv\")\n\n# Select the desired columns (case_id and amount_1115A)\ncredit_bureau_df = credit_bureau_df[['case_id', 'amount_1115A']]\n\n# Display the first few rows of the DataFrame\nprint(credit_bureau_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:39:06.373935Z","iopub.execute_input":"2024-04-06T20:39:06.374378Z","iopub.status.idle":"2024-04-06T20:39:07.240855Z","shell.execute_reply.started":"2024-04-06T20:39:06.374349Z","shell.execute_reply":"2024-04-06T20:39:07.239621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"credit_bureau_df.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:39:23.515469Z","iopub.execute_input":"2024-04-06T20:39:23.515903Z","iopub.status.idle":"2024-04-06T20:39:23.524219Z","shell.execute_reply.started":"2024-04-06T20:39:23.515869Z","shell.execute_reply":"2024-04-06T20:39:23.52282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count zero and non-null values in the 'amount_1115A' column\ncount_zero = (credit_bureau_df['amount_1115A'] == 0).sum()\ncount_non_null = credit_bureau_df['amount_1115A'].count()\n\nprint(\"Total count of zero values:\", count_zero)\nprint(\"Total count of non-null values:\", count_non_null)","metadata":{"execution":{"iopub.status.busy":"2024-04-06T20:40:08.530675Z","iopub.execute_input":"2024-04-06T20:40:08.531195Z","iopub.status.idle":"2024-04-06T20:40:08.542082Z","shell.execute_reply.started":"2024-04-06T20:40:08.531159Z","shell.execute_reply":"2024-04-06T20:40:08.540658Z"},"trusted":true},"execution_count":null,"outputs":[]}]}