{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">HOME CREDIT DATA ANALYSIS</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">IMPORTING VARIOUS MODULES</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.style\n%matplotlib inline\nimport seaborn as sns; sns.set() # for plot styling\nfrom scipy import stats\nplt.rcParams['figure.figsize']=[15,8]\nfrom matplotlib.colors import ListedColormap\nimport scipy.cluster.hierarchy as sch\nfrom scipy.cluster.hierarchy import dendrogram,linkage,fcluster\nfrom sklearn.cluster import AgglomerativeClustering\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.cluster import KMeans\nfrom sklearn.cluster import MiniBatchKMeans\nfrom sklearn.metrics import silhouette_samples\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import classification_report \nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn import metrics,model_selection\nfrom sklearn.preprocessing import scale\nfrom sklearn.decomposition import PCA\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import tree\nfrom scipy.stats import zscore\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:06.537371Z","iopub.execute_input":"2024-05-08T23:36:06.538044Z","iopub.status.idle":"2024-05-08T23:36:10.131234Z","shell.execute_reply.started":"2024-05-08T23:36:06.538011Z","shell.execute_reply":"2024-05-08T23:36:10.129856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">LOADING DATASET</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv')\ntest = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:10.133350Z","iopub.execute_input":"2024-05-08T23:36:10.134003Z","iopub.status.idle":"2024-05-08T23:36:11.435388Z","shell.execute_reply.started":"2024-05-08T23:36:10.133959Z","shell.execute_reply":"2024-05-08T23:36:11.434160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.436664Z","iopub.execute_input":"2024-05-08T23:36:11.437078Z","iopub.status.idle":"2024-05-08T23:36:11.460440Z","shell.execute_reply.started":"2024-05-08T23:36:11.437043Z","shell.execute_reply":"2024-05-08T23:36:11.459272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.463632Z","iopub.execute_input":"2024-05-08T23:36:11.464471Z","iopub.status.idle":"2024-05-08T23:36:11.477578Z","shell.execute_reply.started":"2024-05-08T23:36:11.464428Z","shell.execute_reply":"2024-05-08T23:36:11.476409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.479105Z","iopub.execute_input":"2024-05-08T23:36:11.479550Z","iopub.status.idle":"2024-05-08T23:36:11.584643Z","shell.execute_reply.started":"2024-05-08T23:36:11.479513Z","shell.execute_reply":"2024-05-08T23:36:11.583486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.585851Z","iopub.execute_input":"2024-05-08T23:36:11.586181Z","iopub.status.idle":"2024-05-08T23:36:11.593845Z","shell.execute_reply.started":"2024-05-08T23:36:11.586152Z","shell.execute_reply":"2024-05-08T23:36:11.592310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Summary Statistics of numeric variables:","metadata":{}},{"cell_type":"code","source":"train.describe().round(2).T","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.595148Z","iopub.execute_input":"2024-05-08T23:36:11.595910Z","iopub.status.idle":"2024-05-08T23:36:11.770508Z","shell.execute_reply.started":"2024-05-08T23:36:11.595873Z","shell.execute_reply":"2024-05-08T23:36:11.769390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include=\"all\").T.round(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:11.772239Z","iopub.execute_input":"2024-05-08T23:36:11.772599Z","iopub.status.idle":"2024-05-08T23:36:12.112067Z","shell.execute_reply.started":"2024-05-08T23:36:11.772570Z","shell.execute_reply":"2024-05-08T23:36:12.111176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_vars = train.select_dtypes(include=['int64', 'float64']).columns.tolist()\ncategorical_vars = train.select_dtypes(include=['object']).columns.tolist()                           \nprint('Numerical variables:', numerical_vars)\nprint('Categorical variables:', categorical_vars)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.113579Z","iopub.execute_input":"2024-05-08T23:36:12.114236Z","iopub.status.idle":"2024-05-08T23:36:12.161075Z","shell.execute_reply.started":"2024-05-08T23:36:12.114195Z","shell.execute_reply":"2024-05-08T23:36:12.159817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of categorical and numerical variables\ncategorical_count = train.select_dtypes(include='object').shape[1]\nnumerical_count = train.select_dtypes(exclude='object').shape[1]\n\nprint(f\"Number of categorical variables: {categorical_count}\")\nprint(f\"Number of numerical variables: {numerical_count}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.165820Z","iopub.execute_input":"2024-05-08T23:36:12.166163Z","iopub.status.idle":"2024-05-08T23:36:12.208524Z","shell.execute_reply.started":"2024-05-08T23:36:12.166136Z","shell.execute_reply":"2024-05-08T23:36:12.207392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique values for categorical features\nprint(train.select_dtypes(include=['object']).nunique())","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.209690Z","iopub.execute_input":"2024-05-08T23:36:12.210004Z","iopub.status.idle":"2024-05-08T23:36:12.331129Z","shell.execute_reply.started":"2024-05-08T23:36:12.209978Z","shell.execute_reply":"2024-05-08T23:36:12.329907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Value","metadata":{}},{"cell_type":"code","source":"missing_train =  train.isnull().sum().to_frame().rename(columns={0:\"Total No. of Missing Values\"})\nmissing_train[\"% of Missing Values\"] = round((missing_train[\"Total No. of Missing Values\"]/len( train))*100,2)\nmissing_train","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.332655Z","iopub.execute_input":"2024-05-08T23:36:12.333097Z","iopub.status.idle":"2024-05-08T23:36:12.421768Z","shell.execute_reply.started":"2024-05-08T23:36:12.333068Z","shell.execute_reply":"2024-05-08T23:36:12.420988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the percentage of missing values for each column\nmissing_values_percentage = train.isnull().mean() * 100\n\n# Now you can sort and visualize the missing values\nmissing_values_percentage_sorted = missing_values_percentage.sort_values()\n\n# Visualization code\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(10, 6))\nsns.barplot(x=missing_values_percentage_sorted, y=missing_values_percentage_sorted.index)\nplt.title('Percentage of Missing Values in Each Column (Ascending Order)')\nplt.xlabel('Percentage of Missing Values')\nplt.ylabel('Columns')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.422821Z","iopub.execute_input":"2024-05-08T23:36:12.423649Z","iopub.status.idle":"2024-05-08T23:36:12.841913Z","shell.execute_reply.started":"2024-05-08T23:36:12.423621Z","shell.execute_reply":"2024-05-08T23:36:12.840845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(train.isnull(),cbar=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:12.843340Z","iopub.execute_input":"2024-05-08T23:36:12.843647Z","iopub.status.idle":"2024-05-08T23:36:24.130728Z","shell.execute_reply.started":"2024-05-08T23:36:12.843623Z","shell.execute_reply":"2024-05-08T23:36:24.129499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling missing values\n# Imputing missing values with the mean for continuous variables and mode for categorical variables\nfor col in train.columns:\n    if train[col].dtype == 'object':\n        train[col].fillna(train[col].mode()[0], inplace=True)\n    else:\n        train[col].fillna(train[col].mean(), inplace=True)\n\n# Checking for missing values before imputation\nmissing_values = train.isnull().sum()\n\n# Rechecking for missing values after imputation\nmissing_values_after = train.isnull().sum()\n\n(missing_values, missing_values_after)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:24.132164Z","iopub.execute_input":"2024-05-08T23:36:24.132859Z","iopub.status.idle":"2024-05-08T23:36:24.482512Z","shell.execute_reply.started":"2024-05-08T23:36:24.132813Z","shell.execute_reply":"2024-05-08T23:36:24.481272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_train =  train.isnull().sum().to_frame().rename(columns={0:\"Total No. of Missing Values\"})\nmissing_train[\"% of Missing Values\"] = round((missing_train[\"Total No. of Missing Values\"]/len( train))*100,2)\nmissing_train","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:24.483959Z","iopub.execute_input":"2024-05-08T23:36:24.484717Z","iopub.status.idle":"2024-05-08T23:36:24.573834Z","shell.execute_reply.started":"2024-05-08T23:36:24.484677Z","shell.execute_reply":"2024-05-08T23:36:24.572671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Duplicate Value","metadata":{}},{"cell_type":"code","source":"train[train.duplicated(keep=False)]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:24.575517Z","iopub.execute_input":"2024-05-08T23:36:24.576307Z","iopub.status.idle":"2024-05-08T23:36:24.979084Z","shell.execute_reply.started":"2024-05-08T23:36:24.576277Z","shell.execute_reply":"2024-05-08T23:36:24.977852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:24.980333Z","iopub.execute_input":"2024-05-08T23:36:24.980664Z","iopub.status.idle":"2024-05-08T23:36:25.331468Z","shell.execute_reply.started":"2024-05-08T23:36:24.980637Z","shell.execute_reply":"2024-05-08T23:36:25.330267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop duplicate rows from the DataFrame\ntrain.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:25.332717Z","iopub.execute_input":"2024-05-08T23:36:25.333044Z","iopub.status.idle":"2024-05-08T23:36:25.708376Z","shell.execute_reply.started":"2024-05-08T23:36:25.333018Z","shell.execute_reply":"2024-05-08T23:36:25.707255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:25.710287Z","iopub.execute_input":"2024-05-08T23:36:25.710670Z","iopub.status.idle":"2024-05-08T23:36:25.718097Z","shell.execute_reply.started":"2024-05-08T23:36:25.710641Z","shell.execute_reply":"2024-05-08T23:36:25.716747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the list of categorical columns\ncat_cols = train.select_dtypes(include='object').columns.tolist()\n\n# Create a DataFrame containing counts of unique values for each categorical column\ncat_train = pd.DataFrame(train[cat_cols].melt(var_name='column', value_name='value')\n                      .value_counts()).rename(columns={0: 'count'}).sort_values(by=['column', 'count'])\n\n# Display summary statistics of categorical variables\ndisplay(train[cat_cols].describe())\n\n# Display counts of unique values for each categorical column\ndisplay(cat_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:25.719781Z","iopub.execute_input":"2024-05-08T23:36:25.720189Z","iopub.status.idle":"2024-05-08T23:36:26.386393Z","shell.execute_reply.started":"2024-05-08T23:36:25.720107Z","shell.execute_reply":"2024-05-08T23:36:26.385205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include='O').T","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:26.387941Z","iopub.execute_input":"2024-05-08T23:36:26.388278Z","iopub.status.idle":"2024-05-08T23:36:26.591171Z","shell.execute_reply.started":"2024-05-08T23:36:26.388250Z","shell.execute_reply":"2024-05-08T23:36:26.589991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspect useless features\ntrain.nunique().sort_values()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:26.592673Z","iopub.execute_input":"2024-05-08T23:36:26.593047Z","iopub.status.idle":"2024-05-08T23:36:26.795630Z","shell.execute_reply.started":"2024-05-08T23:36:26.593018Z","shell.execute_reply":"2024-05-08T23:36:26.794406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">EXPLORATOTY DATA ANALYSIS</h2>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## Univariate Analysis","metadata":{}},{"cell_type":"code","source":"# Calculate skewness for numerical columns\nskewness = train.select_dtypes(include=['int64', 'float64']).skew()\n\n# Count the number of numerical columns\nnum_cols_count = len(train.select_dtypes(include=['int64', 'float64']).columns)\n\n# Determine the layout for subplots\nnum_rows = (num_cols_count + 3) // 4  # Adjust the number of columns in each row\nnum_cols = min(4, num_cols_count)  # Maximum of 4 columns in each row\n\n# Plot histograms for numerical columns to visualize distributions and identify anomalies\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 10))\n\nfor i in range(num_rows):\n    for j in range(num_cols):\n        col_idx = i * num_cols + j\n        if col_idx < num_cols_count:\n            col = train.select_dtypes(include=['int64', 'float64']).columns[col_idx]\n            if num_rows == 1:\n                axes[j].hist(train[col], bins=15, color='green', alpha=0.7)\n                axes[j].set_title(f'{col}')\n                axes[j].set_xlabel(col)\n                axes[j].set_ylabel('Frequency')\n                skew_val = skewness[col]\n                axes[j].text(0.5, 0.5, f'Skewness: {skew_val:.2f}', horizontalalignment='center',\n                             verticalalignment='center', transform=axes[j].transAxes, fontsize=10, color='red')\n            else:\n                axes[i, j].hist(train[col], bins=15, color='green', alpha=0.7)\n                axes[i, j].set_title(f'{col}')\n                axes[i, j].set_xlabel(col)\n                axes[i, j].set_ylabel('Frequency')\n                skew_val = skewness[col]\n                axes[i, j].text(0.5, 0.5, f'Skewness: {skew_val:.2f}', horizontalalignment='center',\n                                 verticalalignment='center', transform=axes[i, j].transAxes, fontsize=10, color='red')\n\nplt.tight_layout()\nplt.show()\n\n# Print skewness values\nprint(\"Skewness:\")\nprint(skewness)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:26.797149Z","iopub.execute_input":"2024-05-08T23:36:26.798188Z","iopub.status.idle":"2024-05-08T23:36:28.536953Z","shell.execute_reply.started":"2024-05-08T23:36:26.798152Z","shell.execute_reply":"2024-05-08T23:36:28.535879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the boxplot with rotated text labels\ntrain.plot(kind='box', rot=45,color='green')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:28.538576Z","iopub.execute_input":"2024-05-08T23:36:28.539188Z","iopub.status.idle":"2024-05-08T23:36:29.173728Z","shell.execute_reply.started":"2024-05-08T23:36:28.539149Z","shell.execute_reply":"2024-05-08T23:36:29.172546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter numeric columns\nnumeric_cols = train.select_dtypes(include=['int64', 'float64']).columns\n\n# Plotting boxplots for each numerical feature to identify outliers\nfor column in numeric_cols:\n    plt.figure(figsize=(10, 6))\n    sns.boxplot(x=train[column],palette='rainbow')\n    plt.title(f'Boxplot of {column}')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:29.175182Z","iopub.execute_input":"2024-05-08T23:36:29.175523Z","iopub.status.idle":"2024-05-08T23:36:30.634805Z","shell.execute_reply.started":"2024-05-08T23:36:29.175497Z","shell.execute_reply":"2024-05-08T23:36:30.633506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multivariate Analysis","metadata":{}},{"cell_type":"code","source":"# Correlation matrix\n\n# Select only the numeric columns from the DataFrame\nnumeric_columns = train.select_dtypes(include=['number'])\n\n# Calculate the correlation matrix\ncorrelation_matrix = numeric_columns.corr()\n\n# Create a heatmap to visualize the correlations\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap='rainbow', fmt=\".2f\", linewidths=0.5)\nplt.title('Correlation Matrix')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:30.636204Z","iopub.execute_input":"2024-05-08T23:36:30.636667Z","iopub.status.idle":"2024-05-08T23:36:31.212415Z","shell.execute_reply.started":"2024-05-08T23:36:30.636627Z","shell.execute_reply":"2024-05-08T23:36:31.211147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Heatmap Plotting\n# Select only numeric columns\nnumeric_columns = train.select_dtypes(include=['int64', 'float64'])\n\n# Calculate the correlation matrix\ncorr_matrix = numeric_columns.corr()\n\n# Filter correlation matrix to include values greater than 0.5 or less than -0.5\ncorr_matrix_filtered = corr_matrix[(corr_matrix > 0.5) | (corr_matrix < -0.5)]\n\n# Plot the heatmap with filtered correlation values\nplt.figure(figsize=(12, 10))\nsns.heatmap(corr_matrix_filtered, annot=True, cmap='rainbow', fmt=\".2f\", linewidths=0.5)\nplt.title('Correlation Heatmap of Numeric Features (|Correlation| > 0.5)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:31.219632Z","iopub.execute_input":"2024-05-08T23:36:31.220042Z","iopub.status.idle":"2024-05-08T23:36:31.781189Z","shell.execute_reply.started":"2024-05-08T23:36:31.220008Z","shell.execute_reply":"2024-05-08T23:36:31.780015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explore categorical features\nfor column in train.select_dtypes(include=['object']):\n    sns.countplot(x=column, data=train,palette='rainbow')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:31.783234Z","iopub.execute_input":"2024-05-08T23:36:31.784097Z","iopub.status.idle":"2024-05-08T23:36:41.829906Z","shell.execute_reply.started":"2024-05-08T23:36:31.784062Z","shell.execute_reply":"2024-05-08T23:36:41.828403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explore categorical features\nfor column in train.select_dtypes(include=['object']):\n    plt.figure(figsize=(10, 6))\n    ax = sns.countplot(x=column, data=train,palette='rainbow')\n    \n    # Add count and percentage annotations to each bar\n    total = len(train[column])\n    for p in ax.patches:\n        percentage = '{:.1f}%'.format(100 * p.get_height() / total)\n        count = p.get_height()\n        x = p.get_x() + p.get_width() / 2\n        y = p.get_height()\n        ax.annotate(f'{count}\\n{percentage}', (x, y), ha='center', va='bottom')\n    \n    plt.title(f'Count Plot for {column}', fontsize=15)\n    plt.xlabel(column, fontsize=12)\n    plt.ylabel('Count', fontsize=12)\n    plt.xticks(rotation=45)  # Rotate x-axis labels for better readability\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:41.831634Z","iopub.execute_input":"2024-05-08T23:36:41.832566Z","iopub.status.idle":"2024-05-08T23:36:55.152369Z","shell.execute_reply.started":"2024-05-08T23:36:41.832520Z","shell.execute_reply":"2024-05-08T23:36:55.151225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Sample DataFrame (replace this with your actual DataFrame)\n# train = ...\n\n# Function to remove outliers using the IQR method\ndef remove_outliers_iqr(train):\n    # Select only numeric columns\n    numeric_train = train.select_dtypes(include=['int64', 'float64'])\n    \n    # Calculate the first quartile (Q1) and third quartile (Q3)\n    Q1 = numeric_train.quantile(0.25)\n    Q3 = numeric_train.quantile(0.75)\n    \n    # Interquartile range (IQR)\n    IQR = Q3 - Q1\n    \n    # Define the lower and upper bounds for outlier detection\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    # Identify outliers\n    outliers = ((numeric_train < lower_bound) | (numeric_train > upper_bound)).any(axis=1)\n    \n    # Count the number of outliers removed\n    num_outliers_removed = outliers.sum()\n    \n    # Filter DataFrame based on rows without outliers\n    train_no_outliers = train[~outliers]\n    \n    return train_no_outliers, num_outliers_removed\n\n# Remove outliers using IQR method and get the number of outliers removed\ntrain_no_outliers, num_outliers_removed = remove_outliers_iqr(train)\n\nprint(\"Number of outliers removed:\", num_outliers_removed)\n\n# Function to plot boxplots before and after removing outliers\ndef plot_boxplots_before_after(train_before, train_after):\n    # Set up the figure\n    fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(12, 6))\n    \n    # Boxplot before removing outliers (blue color)\n    sns.boxplot(data=train_before, ax=axes[0], color='blue')\n    axes[0].set_title('Before Removing Outliers')\n    \n    # Boxplot after removing outliers (green color)\n    sns.boxplot(data=train_after, ax=axes[1], color='green')\n    axes[1].set_title('After Removing Outliers')\n    \n    # Adjust layout\n    plt.tight_layout()\n    plt.show()\n\n# Plot boxplots before and after outlier removal\nplot_boxplots_before_after(train, train_no_outliers)\nprint(\"Number of outliers removed:\", num_outliers_removed)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:55.153754Z","iopub.execute_input":"2024-05-08T23:36:55.154102Z","iopub.status.idle":"2024-05-08T23:36:56.909515Z","shell.execute_reply.started":"2024-05-08T23:36:55.154071Z","shell.execute_reply":"2024-05-08T23:36:56.908345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">LABEL ENCODING</h2>\n</div>","metadata":{}},{"cell_type":"markdown","source":"### Segregate Categorical and Numerical Columns","metadata":{}},{"cell_type":"code","source":"catcol = []\nnumcol = []\n\nfor col in train.columns:\n    if train[col].dtype == 'object':\n        catcol.append(col)\n    else:\n        numcol.append(col)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:56.911117Z","iopub.execute_input":"2024-05-08T23:36:56.911747Z","iopub.status.idle":"2024-05-08T23:36:56.917552Z","shell.execute_reply.started":"2024-05-08T23:36:56.911716Z","shell.execute_reply":"2024-05-08T23:36:56.916304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Categorical Columns:\",catcol)\nprint(\"Numerical Columns:\", numcol)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:56.918913Z","iopub.execute_input":"2024-05-08T23:36:56.919366Z","iopub.status.idle":"2024-05-08T23:36:56.932351Z","shell.execute_reply.started":"2024-05-08T23:36:56.919307Z","shell.execute_reply":"2024-05-08T23:36:56.931067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nencoder = LabelEncoder()\n\nfor col in catcol:\n    train[col] = encoder.fit_transform(train[col])","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:56.933735Z","iopub.execute_input":"2024-05-08T23:36:56.934177Z","iopub.status.idle":"2024-05-08T23:36:57.242180Z","shell.execute_reply.started":"2024-05-08T23:36:56.934139Z","shell.execute_reply":"2024-05-08T23:36:57.240720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.245188Z","iopub.execute_input":"2024-05-08T23:36:57.245549Z","iopub.status.idle":"2024-05-08T23:36:57.256412Z","shell.execute_reply.started":"2024-05-08T23:36:57.245521Z","shell.execute_reply":"2024-05-08T23:36:57.255383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">DATA SCALING</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscale = MinMaxScaler()\n\nfor col in numcol:\n    train[[col]] = scale.fit_transform(train[[col]])","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.257950Z","iopub.execute_input":"2024-05-08T23:36:57.258537Z","iopub.status.idle":"2024-05-08T23:36:57.356719Z","shell.execute_reply.started":"2024-05-08T23:36:57.258506Z","shell.execute_reply":"2024-05-08T23:36:57.355642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.358027Z","iopub.execute_input":"2024-05-08T23:36:57.358398Z","iopub.status.idle":"2024-05-08T23:36:57.372261Z","shell.execute_reply.started":"2024-05-08T23:36:57.358366Z","shell.execute_reply":"2024-05-08T23:36:57.370896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train.drop('target',axis=1)\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.373718Z","iopub.execute_input":"2024-05-08T23:36:57.374026Z","iopub.status.idle":"2024-05-08T23:36:57.410260Z","shell.execute_reply.started":"2024-05-08T23:36:57.374002Z","shell.execute_reply":"2024-05-08T23:36:57.409255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">TRAIN & TEST SPLIT</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(x,y,test_size = 0.15,shuffle = True,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.411825Z","iopub.execute_input":"2024-05-08T23:36:57.412172Z","iopub.status.idle":"2024-05-08T23:36:57.663170Z","shell.execute_reply.started":"2024-05-08T23:36:57.412143Z","shell.execute_reply":"2024-05-08T23:36:57.662298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the shapes of the resulting sets\nprint(\"Training set shape:\", x_train.shape, y_train.shape)\nprint(\"Testing set shape:\", x_test.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.664682Z","iopub.execute_input":"2024-05-08T23:36:57.665390Z","iopub.status.idle":"2024-05-08T23:36:57.670696Z","shell.execute_reply.started":"2024-05-08T23:36:57.665357Z","shell.execute_reply":"2024-05-08T23:36:57.669833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"text-align: center; background-color: #8ABEB9; padding: 10px;\">\n    <h2 style=\"font-weight: bold;\">MODEL BUILDING</h2>\n</div>","metadata":{}},{"cell_type":"code","source":"def evaluate_model(true,predicted):\n    mse = mean_squared_error(true, predicted)\n    mae = mean_absolute_error(true,predicted)\n    rmse = np.sqrt(mse)\n    r2_square = r2_score(true,predicted)\n    return mae,rmse,r2_square","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.672145Z","iopub.execute_input":"2024-05-08T23:36:57.672782Z","iopub.status.idle":"2024-05-08T23:36:57.680502Z","shell.execute_reply.started":"2024-05-08T23:36:57.672753Z","shell.execute_reply":"2024-05-08T23:36:57.679631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\n\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor, AdaBoostRegressor\nfrom xgboost import XGBRegressor\n\nmodels = {\n    \"Linear Regression\": LinearRegression(),\n    \"Lasso\": Lasso(),\n    \"Ridge\": Ridge(),\n    \"k-Neighbors Regression\": KNeighborsRegressor(),\n    \"Decision Tree\": DecisionTreeRegressor(),\n    \"Random Forest Regressor\": RandomForestRegressor(n_estimators=100, random_state=0),\n    \"AdaBoost Regressor\": AdaBoostRegressor(),\n    \"XGBRegressor\": XGBRegressor()\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.682007Z","iopub.execute_input":"2024-05-08T23:36:57.682678Z","iopub.status.idle":"2024-05-08T23:36:57.871059Z","shell.execute_reply.started":"2024-05-08T23:36:57.682624Z","shell.execute_reply":"2024-05-08T23:36:57.869906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_text=\"\"\nfor i in range(len(models)):\n            model = list(models.values())[i]\n            model.fit(x_train,y_train)\n\n            #Make prediction:\n            y_train_pred = model.predict(x_train)\n            y_test_pred = model.predict(x_test)\n\n            #Evaluate Train and Test dataset :\n\n            model_train_mae, model_train_rmse, model_train_r2 = evaluate_model(y_train, y_train_pred)\n            model_test_mae, model_test_rmse, model_test_r2 = evaluate_model(y_test, y_test_pred)\n            model_name = list(models.keys())[i]\n           \n            print(model_name)\n\n            print(\"Model Performance for Training set :\")\n\n            print('Root Mean Squared Error :',model_train_rmse)\n            print(\"Mean Absolute Error : \", model_train_mae)\n            print(\"R2 Score : \", model_train_r2)\n\n            print(\"----------------------------------------------------\")\n            \n            print(\"Model Performance for Testing set :\")\n\n            print('Root Mean Squared Error : ', model_test_rmse)\n            print('Mean Absolute Error :  ',{model_test_mae})\n            print('R2 Score : ', {model_test_r2})\n            print()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T23:36:57.872534Z","iopub.execute_input":"2024-05-08T23:36:57.872884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These metrics provide insights into the performance of each model. While some models perform well on the training set, but failing to unseen data, as indicated by their performance on the testing set.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}