{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport missingno as msno\n\n\nimport warnings\n\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-23T13:39:48.555276Z","iopub.execute_input":"2024-09-23T13:39:48.555716Z","iopub.status.idle":"2024-09-23T13:39:48.563302Z","shell.execute_reply.started":"2024-09-23T13:39:48.555671Z","shell.execute_reply":"2024-09-23T13:39:48.561793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data \ntrain = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:12:43.761793Z","iopub.execute_input":"2024-09-23T13:12:43.762217Z","iopub.status.idle":"2024-09-23T13:12:43.856076Z","shell.execute_reply.started":"2024-09-23T13:12:43.762176Z","shell.execute_reply":"2024-09-23T13:12:43.854787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rename columns to lowercase without symbols\ntrain.columns = train.columns.str.lower().str.replace(r'\\W+', '_', regex=True)\ntest.columns = test.columns.str.lower().str.replace(r'\\W+', '_', regex=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:14:54.363579Z","iopub.execute_input":"2024-09-23T13:14:54.364015Z","iopub.status.idle":"2024-09-23T13:14:54.374913Z","shell.execute_reply.started":"2024-09-23T13:14:54.363974Z","shell.execute_reply":"2024-09-23T13:14:54.373512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Features","metadata":{}},{"cell_type":"code","source":"target_columns = set(train.columns) - set(test.columns)\ntarget_columns","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:14:57.365287Z","iopub.execute_input":"2024-09-23T13:14:57.365793Z","iopub.status.idle":"2024-09-23T13:14:57.375153Z","shell.execute_reply.started":"2024-09-23T13:14:57.365745Z","shell.execute_reply":"2024-09-23T13:14:57.373712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = set(train.columns) - target_columns\nfeatures","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:16:02.257348Z","iopub.execute_input":"2024-09-23T13:16:02.257842Z","iopub.status.idle":"2024-09-23T13:16:02.267026Z","shell.execute_reply.started":"2024-09-23T13:16:02.257795Z","shell.execute_reply":"2024-09-23T13:16:02.265689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Target distribution","metadata":{}},{"cell_type":"code","source":"# Target distribution visualization\nplt.figure(figsize=(8, 6))\nsns.countplot(x='sii', data=train, palette='Set2')\nplt.title('Target (sii) Distribution')\nplt.xlabel('sii')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:19:13.293818Z","iopub.execute_input":"2024-09-23T13:19:13.294296Z","iopub.status.idle":"2024-09-23T13:19:13.639551Z","shell.execute_reply.started":"2024-09-23T13:19:13.294249Z","shell.execute_reply":"2024-09-23T13:19:13.638070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`Key Insights:`\n\nClass Imbalance:\n\nThe \"None\" category (0.0) is the most prevalent class, followed by \"Mild\" (1.0) and \"Moderate\" (2.0). The \"Severe\" (3.0) class is significantly underrepresented, which could affect the model's ability to learn this class effectively.\nImbalanced Dataset:\n\nSince there is a large disparity between the number of samples in each category, especially for the \"Severe\" class, you may need to apply techniques like:\n* Resampling: Either oversample the minority classes (e.g., 3.0) or undersample the majority class (0.0).\n* Class Weights: Set class weights in your model to give higher importance to the underrepresented classes during training.\n* Synthetic Data Generation: Techniques such as SMOTE (Synthetic Minority Over-sampling Technique) could be used to generate synthetic samples for the minority class.","metadata":{}},{"cell_type":"code","source":"# Target distribution visualization\nplt.figure(figsize=(8, 6))\nsns.distplot(train['pciat_pciat_total'])\nplt.title('Target (pciat_pciat_total) Distribution')\nplt.xlabel('pciat_pciat_total')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:16:33.962737Z","iopub.execute_input":"2024-09-23T14:16:33.963978Z","iopub.status.idle":"2024-09-23T14:16:34.364591Z","shell.execute_reply.started":"2024-09-23T14:16:33.963922Z","shell.execute_reply":"2024-09-23T14:16:34.362909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"train[list(features)].isnull().sum() / len(train) * 100","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:57:55.531562Z","iopub.execute_input":"2024-09-23T13:57:55.532113Z","iopub.status.idle":"2024-09-23T13:57:55.561356Z","shell.execute_reply.started":"2024-09-23T13:57:55.532067Z","shell.execute_reply":"2024-09-23T13:57:55.560121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[list(target_columns)].isnull().sum() / len(train) * 100","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:59:01.493090Z","iopub.execute_input":"2024-09-23T13:59:01.493571Z","iopub.status.idle":"2024-09-23T13:59:01.509319Z","shell.execute_reply.started":"2024-09-23T13:59:01.493501Z","shell.execute_reply":"2024-09-23T13:59:01.507939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing values visualization\nplt.figure(figsize=(10, 6))\nmsno.bar(test)\nplt.title('Missing Values per Column')\nplt.show()\n\n# Heatmap of missing values\nplt.figure(figsize=(10, 6))\nmsno.heatmap(test)\nplt.title('Missing Values Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:32:35.053878Z","iopub.execute_input":"2024-09-23T13:32:35.054313Z","iopub.status.idle":"2024-09-23T13:32:42.464954Z","shell.execute_reply.started":"2024-09-23T13:32:35.054275Z","shell.execute_reply":"2024-09-23T13:32:42.463623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(test)\nplt.title('Missing Data Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:34:04.219711Z","iopub.execute_input":"2024-09-23T13:34:04.220241Z","iopub.status.idle":"2024-09-23T13:34:04.826464Z","shell.execute_reply.started":"2024-09-23T13:34:04.220194Z","shell.execute_reply":"2024-09-23T13:34:04.825207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"code","source":"# Compute the correlation matrix for numeric columns\ncorr_matrix = train.select_dtypes(include='number').corr()\ncorr_matrix","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:39:53.693576Z","iopub.execute_input":"2024-09-23T13:39:53.694110Z","iopub.status.idle":"2024-09-23T13:39:53.840017Z","shell.execute_reply.started":"2024-09-23T13:39:53.694057Z","shell.execute_reply":"2024-09-23T13:39:53.838731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 10))\n# Create a heatmap for the correlation matrix\nsns.heatmap(corr_matrix, annot=False, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title('Correlation Matrix Heatmap', fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:38:16.210715Z","iopub.execute_input":"2024-09-23T13:38:16.211164Z","iopub.status.idle":"2024-09-23T13:38:17.398013Z","shell.execute_reply.started":"2024-09-23T13:38:16.211122Z","shell.execute_reply":"2024-09-23T13:38:17.396493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for target_col in target_columns :\n    if target_col!='pciat_season':\n        # Drop the target column from correlation values and sort by absolute correlation\n        target_corr = corr_matrix[target_col].drop(target_columns, errors='ignore').sort_values(ascending=False, key=abs)\n\n        # Plot the correlation as a barplot\n        plt.figure(figsize=(10, 6))\n        sns.barplot(x=target_corr.values, y=target_corr.index, palette=\"coolwarm\")\n        plt.title(f'Correlation of Features with {target_col}')\n        plt.xlabel('Correlation Coefficient')\n        plt.ylabel('Feature')\n        \n        plt.axvline(x=0.1, color='red', linestyle='--', label='Threshold 0.1')\n        plt.axvline(x=-0.1, color='red', linestyle='--', label='Threshold 0.1')\n        plt.legend()\n        \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:48:39.536780Z","iopub.execute_input":"2024-09-23T13:48:39.537218Z","iopub.status.idle":"2024-09-23T13:49:03.008256Z","shell.execute_reply.started":"2024-09-23T13:48:39.537178Z","shell.execute_reply":"2024-09-23T13:49:03.006936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distributions","metadata":{}},{"cell_type":"markdown","source":"### Categorical","metadata":{}},{"cell_type":"code","source":"# Set a threshold for the number of unique values to consider for categorical columns\nnunique_threshold = 12  # Adjust the threshold as per your data (e.g., 20 unique values)\n\n# Select columns with 'object' type or columns with less than `nunique_threshold` unique values\ncategorical_columns = [col for col in train.columns if train[col].nunique() < nunique_threshold]","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:54:28.756544Z","iopub.execute_input":"2024-09-23T13:54:28.757014Z","iopub.status.idle":"2024-09-23T13:54:28.782852Z","shell.execute_reply.started":"2024-09-23T13:54:28.756966Z","shell.execute_reply":"2024-09-23T13:54:28.781609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = set(categorical_columns) - target_columns","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:54:46.350587Z","iopub.execute_input":"2024-09-23T13:54:46.351956Z","iopub.status.idle":"2024-09-23T13:54:46.358605Z","shell.execute_reply.started":"2024-09-23T13:54:46.351882Z","shell.execute_reply":"2024-09-23T13:54:46.356876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[list(categorical_columns)] = train[list(categorical_columns)].fillna('missing')","metadata":{"execution":{"iopub.status.busy":"2024-09-23T13:56:01.887516Z","iopub.execute_input":"2024-09-23T13:56:01.888023Z","iopub.status.idle":"2024-09-23T13:56:01.918761Z","shell.execute_reply.started":"2024-09-23T13:56:01.887978Z","shell.execute_reply":"2024-09-23T13:56:01.917372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in categorical_columns:\n    # Target distribution visualization\n    plt.figure(figsize=(8, 6))\n    sns.countplot(x=col, data=train, palette='Set2')\n    plt.title(f'{col} Distribution')\n    plt.xlabel(col)\n    plt.ylabel('Count')\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:01:14.438686Z","iopub.execute_input":"2024-09-23T14:01:14.439150Z","iopub.status.idle":"2024-09-23T14:01:19.371895Z","shell.execute_reply.started":"2024-09-23T14:01:14.439107Z","shell.execute_reply":"2024-09-23T14:01:19.370549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Numeric","metadata":{}},{"cell_type":"code","source":"# Iterate through each numeric column and create a boxplot for target 'sii'\nfor num_col in features - categorical_columns:\n    if num_col != 'sii':  # Ensure target column itself is not plotted\n        plt.figure(figsize=(10, 6))\n        sns.boxplot(x='sii', y=num_col, data=train, palette='Set2')\n        plt.title(f'Boxplot of {num_col} by sii')\n        plt.xlabel('sii')\n        plt.ylabel(num_col)\n        plt.xticks(rotation=45)  # Rotate x-ticks if needed\n        plt.tight_layout()  # Adjust layout to prevent label overlap\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:07:41.031673Z","iopub.execute_input":"2024-09-23T14:07:41.032151Z","iopub.status.idle":"2024-09-23T14:09:09.854137Z","shell.execute_reply.started":"2024-09-23T14:07:41.032105Z","shell.execute_reply":"2024-09-23T14:09:09.852798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* lot of outliers (let's try to clip it) ","metadata":{}},{"cell_type":"code","source":"# Function to clip outliers using IQR\ndef clip_outliers(df, col):\n    Q1 = df[col].quantile(0.25)\n    Q3 = df[col].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    return df[col].clip(lower=lower_bound, upper=upper_bound)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:12:54.912161Z","iopub.execute_input":"2024-09-23T14:12:54.912689Z","iopub.status.idle":"2024-09-23T14:12:54.920483Z","shell.execute_reply.started":"2024-09-23T14:12:54.912636Z","shell.execute_reply":"2024-09-23T14:12:54.919145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through each numeric column and create a boxplot for target 'sii'\nfor num_col in train[list(features)].select_dtypes(include='number').columns:\n    if num_col != 'sii':  # Ensure target column itself is not plotted\n        train[num_col] = clip_outliers(train, num_col)\n        plt.figure(figsize=(10, 6))\n        sns.boxplot(x='sii', y=num_col, data=train, palette='Set2')\n        plt.title(f'Boxplot of {num_col} by sii')\n        plt.xlabel('sii')\n        plt.ylabel(num_col)\n        plt.xticks(rotation=45)  # Rotate x-ticks if needed\n        plt.tight_layout()  # Adjust layout to prevent label overlap\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:14:34.078822Z","iopub.execute_input":"2024-09-23T14:14:34.079240Z","iopub.status.idle":"2024-09-23T14:14:48.426107Z","shell.execute_reply.started":"2024-09-23T14:14:34.079201Z","shell.execute_reply":"2024-09-23T14:14:48.424635Z"},"trusted":true},"execution_count":null,"outputs":[]}]}