{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-30T17:08:01.384567Z","iopub.execute_input":"2024-09-30T17:08:01.385456Z","iopub.status.idle":"2024-09-30T17:08:04.247406Z","shell.execute_reply.started":"2024-09-30T17:08:01.385410Z","shell.execute_reply":"2024-09-30T17:08:04.246171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:09:56.835056Z","iopub.execute_input":"2024-09-30T17:09:56.835490Z","iopub.status.idle":"2024-09-30T17:09:56.916513Z","shell.execute_reply.started":"2024-09-30T17:09:56.835449Z","shell.execute_reply":"2024-09-30T17:09:56.915382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:10:13.445236Z","iopub.execute_input":"2024-09-30T17:10:13.445694Z","iopub.status.idle":"2024-09-30T17:10:13.490153Z","shell.execute_reply.started":"2024-09-30T17:10:13.445653Z","shell.execute_reply":"2024-09-30T17:10:13.489057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic Information and Structure","metadata":{}},{"cell_type":"code","source":"print(\"Shape of the dataset:\", df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:10:33.199084Z","iopub.execute_input":"2024-09-30T17:10:33.199535Z","iopub.status.idle":"2024-09-30T17:10:33.206073Z","shell.execute_reply.started":"2024-09-30T17:10:33.199492Z","shell.execute_reply":"2024-09-30T17:10:33.204878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nData types of each column:\\n\", df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:10:49.317433Z","iopub.execute_input":"2024-09-30T17:10:49.317889Z","iopub.status.idle":"2024-09-30T17:10:49.326090Z","shell.execute_reply.started":"2024-09-30T17:10:49.317844Z","shell.execute_reply":"2024-09-30T17:10:49.324904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handling Missing Data","metadata":{}},{"cell_type":"code","source":"missing_values = df.isnull().sum()\nprint(\"\\nMissing values in each column:\\n\", missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:11:19.416956Z","iopub.execute_input":"2024-09-30T17:11:19.417403Z","iopub.status.idle":"2024-09-30T17:11:19.431937Z","shell.execute_reply.started":"2024-09-30T17:11:19.417359Z","shell.execute_reply":"2024-09-30T17:11:19.430724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cleaned = df.dropna(subset=['sii'])\n\n# Filling missing data for numerical columns with the mean\nfor column in df_cleaned.select_dtypes(include=[np.number]).columns:\n    df_cleaned[column] = df_cleaned[column].fillna(df_cleaned[column].mean())\n\n# Filling missing data for categorical columns with the mode\nfor column in df_cleaned.select_dtypes(include=['object', 'category']).columns:\n    df_cleaned[column] = df_cleaned[column].fillna(df_cleaned[column].mode()[0])\n\nprint(\"\\nMissing values after handling:\\n\", df_cleaned.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:30:32.748175Z","iopub.execute_input":"2024-09-30T17:30:32.748606Z","iopub.status.idle":"2024-09-30T17:30:32.804467Z","shell.execute_reply.started":"2024-09-30T17:30:32.748565Z","shell.execute_reply":"2024-09-30T17:30:32.803424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nSummary statistics of numerical columns:\\n\", df.describe())","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:20:14.136550Z","iopub.execute_input":"2024-09-30T17:20:14.137959Z","iopub.status.idle":"2024-09-30T17:20:14.296232Z","shell.execute_reply.started":"2024-09-30T17:20:14.137890Z","shell.execute_reply":"2024-09-30T17:20:14.295153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nSummary statistics of categorical columns:\\n\", df.describe(include=['object', 'category']))","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:20:27.720815Z","iopub.execute_input":"2024-09-30T17:20:27.721260Z","iopub.status.idle":"2024-09-30T17:20:27.766702Z","shell.execute_reply.started":"2024-09-30T17:20:27.721220Z","shell.execute_reply":"2024-09-30T17:20:27.765600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing distributions of numerical features\nnumerical_vars = ['Basic_Demos-Age', 'Physical-BMI', 'PCIAT-PCIAT_Total', 'PreInt_EduHx-computerinternet_hoursday']\n\ndf[numerical_vars] = df[numerical_vars].replace([np.inf, -np.inf], np.nan)\n\nplt.figure(figsize=(15, 5))\n\n# Iterate through the numerical variables and plot their distributions\nfor i, column in enumerate(numerical_vars, 1):\n    plt.subplot(1, len(numerical_vars), i)\n    sns.histplot(df[column], kde=True)\n    plt.title(f'Distribution of {column}')\n\nplt.tight_layout() # Adjust layout and display the plots\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:36:35.364431Z","iopub.execute_input":"2024-09-30T17:36:35.364874Z","iopub.status.idle":"2024-09-30T17:36:37.043536Z","shell.execute_reply.started":"2024-09-30T17:36:35.364833Z","shell.execute_reply":"2024-09-30T17:36:37.042354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count plots for categorical variables\ncategorical_vars = ['Basic_Demos-Sex']  # Add more categorical vars if needed\nfor column in categorical_vars:\n    plt.figure(figsize=(5, 5))\n    sns.countplot(data=df, x=column)\n    plt.title(f'Count plot of {column}')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:21:50.958081Z","iopub.execute_input":"2024-09-30T17:21:50.958795Z","iopub.status.idle":"2024-09-30T17:21:51.194293Z","shell.execute_reply.started":"2024-09-30T17:21:50.958735Z","shell.execute_reply":"2024-09-30T17:21:51.193122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Correlation Analysis**","metadata":{}},{"cell_type":"code","source":"# Calculating correlation between numerical variables\n# Convert relevant columns to numeric type, handling errors\nfor column in df.columns:\n    try:\n        df[column] = pd.to_numeric(df[column], errors='coerce')\n    except:\n        pass\n\ncorrelation_matrix = df.corr()\nprint(\"\\nCorrelation matrix:\\n\", correlation_matrix)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:24:30.964169Z","iopub.execute_input":"2024-09-30T17:24:30.964598Z","iopub.status.idle":"2024-09-30T17:24:31.140132Z","shell.execute_reply.started":"2024-09-30T17:24:30.964560Z","shell.execute_reply":"2024-09-30T17:24:31.138913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing correlations using a heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=False, fmt=\".2f\", cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()\n\n#It can be enhanced using display annotations, adjust color map or masking upper triangle (since it's symmentric)","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:42:30.207506Z","iopub.execute_input":"2024-09-30T17:42:30.207986Z","iopub.status.idle":"2024-09-30T17:42:31.331028Z","shell.execute_reply.started":"2024-09-30T17:42:30.207942Z","shell.execute_reply":"2024-09-30T17:42:31.329812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Outlier Detection","metadata":{}},{"cell_type":"code","source":" # Function to calculate outliers using IQR\ndef count_outliers(data):\n    outlier_counts = {}\n    for column in data.select_dtypes(include=[np.number]).columns:  # Only numeric columns\n        Q1 = data[column].quantile(0.25)\n        Q3 = data[column].quantile(0.75)\n        IQR = Q3 - Q1\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n        outlier_count = ((data[column] < lower_bound) | (data[column] > upper_bound)).sum()\n        outlier_counts[column] = outlier_count\n    return outlier_counts\n\n# Count outliers in each feature\noutlier_counts = count_outliers(df)\n\n# Select the top 10 features with the most outliers\ntop_10_outliers = dict(sorted(outlier_counts.items(), key=lambda item: item[1], reverse=True)[:10])\n\n# Plotting the number of outliers in the top 10 features\nplt.figure(figsize=(12, 6))\nsns.barplot(x=list(top_10_outliers.keys()), y=list(top_10_outliers.values()), palette='viridis')\nplt.title('Features with Most Outliers')\nplt.xlabel('Features')\nplt.ylabel('Number of Outliers')\nplt.xticks(rotation=45, fontsize=10)  # Set fontsize to a smaller value\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:53:55.268579Z","iopub.execute_input":"2024-09-30T17:53:55.269044Z","iopub.status.idle":"2024-09-30T17:53:55.888944Z","shell.execute_reply.started":"2024-09-30T17:53:55.269001Z","shell.execute_reply":"2024-09-30T17:53:55.887906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Relationship","metadata":{}},{"cell_type":"code","source":"# Define the features to analyze relationships\nfeatures_to_analyze = ['Basic_Demos-Age', 'Physical-BMI', 'PCIAT-PCIAT_Total', 'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n# Scatter plots for feature relationships\nplt.figure(figsize=(15, 10))\n\n# Scatter plot between 'Basic_Demos-Age' and 'sii'\nplt.subplot(2, 2, 1)\nsns.scatterplot(data=df, x='Basic_Demos-Age', y='sii')\nplt.title('Scatter Plot: Age vs. Severity Impairment Index')\n\n# Scatter plot between 'Physical-BMI' and 'sii'\nplt.subplot(2, 2, 2)\nsns.scatterplot(data=df, x='Physical-BMI', y='sii')\nplt.title('Scatter Plot: BMI vs. Severity Impairment Index')\n\n# Scatter plot between 'PCIAT-PCIAT_Total' and 'sii'\nplt.subplot(2, 2, 3)\nsns.scatterplot(data=df, x='PCIAT-PCIAT_Total', y='sii')\nplt.title('Scatter Plot: PCIAT Total vs. Severity Impairment Index')\n\n# Scatter plot between 'PreInt_EduHx-computerinternet_hoursday' and 'sii'\nplt.subplot(2, 2, 4)\nsns.scatterplot(data=df, x='PreInt_EduHx-computerinternet_hoursday', y='sii')\nplt.title('Scatter Plot: Internet Hours vs. Severity Impairment Index')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-30T17:54:40.064598Z","iopub.execute_input":"2024-09-30T17:54:40.065064Z","iopub.status.idle":"2024-09-30T17:54:41.740519Z","shell.execute_reply.started":"2024-09-30T17:54:40.065022Z","shell.execute_reply":"2024-09-30T17:54:41.739405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Variable Analysis","metadata":{}},{"cell_type":"code","source":"# Checking the distribution of the target variable (sii)\nplt.figure(figsize=(8, 5))\nsns.countplot(data=df, x='sii')\nplt.title('Distribution of Severity Impairment Index (sii)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T18:01:10.777327Z","iopub.execute_input":"2024-09-30T18:01:10.778635Z","iopub.status.idle":"2024-09-30T18:01:11.006899Z","shell.execute_reply.started":"2024-09-30T18:01:10.778586Z","shell.execute_reply":"2024-09-30T18:01:11.005796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Analyzing how features affect the target variable\nplt.figure(figsize=(10, 6))\n# Replace 'Basic_Demos-Age' with the desired numerical feature from your dataset.\nsns.boxplot(data=df, x='sii', y='Basic_Demos-Age')\nplt.title('Effect of Basic_Demos-Age on Severity Impairment Index (sii)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-30T18:01:11.068155Z","iopub.execute_input":"2024-09-30T18:01:11.068557Z","iopub.status.idle":"2024-09-30T18:01:11.353287Z","shell.execute_reply.started":"2024-09-30T18:01:11.068521Z","shell.execute_reply":"2024-09-30T18:01:11.352176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}