{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt \nimport seaborn as sns\n\nimport warnings\n\nwarnings.filterwarnings('ignore')\nsns.set(style=\"whitegrid\")\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-22T03:51:35.762724Z","iopub.execute_input":"2024-10-22T03:51:35.763145Z","iopub.status.idle":"2024-10-22T03:51:37.048531Z","shell.execute_reply.started":"2024-10-22T03:51:35.763107Z","shell.execute_reply":"2024-10-22T03:51:37.047001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:51:53.629488Z","iopub.execute_input":"2024-10-22T03:51:53.629917Z","iopub.status.idle":"2024-10-22T03:51:53.791624Z","shell.execute_reply.started":"2024-10-22T03:51:53.629878Z","shell.execute_reply":"2024-10-22T03:51:53.790144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{"execution":{"iopub.status.busy":"2024-10-16T00:06:57.272757Z","iopub.execute_input":"2024-10-16T00:06:57.273994Z","iopub.status.idle":"2024-10-16T00:06:57.279541Z","shell.execute_reply.started":"2024-10-16T00:06:57.273938Z","shell.execute_reply":"2024-10-16T00:06:57.27806Z"}}},{"cell_type":"code","source":"train.columns = train.columns.str.lower().str.replace(r'\\W+', '_', regex=True)\ntest.columns = test.columns.str.lower().str.replace(r'\\W+', '_', regex=True)\ntrain.describe().transpose()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:52:42.250644Z","iopub.execute_input":"2024-10-22T03:52:42.251094Z","iopub.status.idle":"2024-10-22T03:52:42.448745Z","shell.execute_reply.started":"2024-10-22T03:52:42.251049Z","shell.execute_reply":"2024-10-22T03:52:42.447535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:53:02.963483Z","iopub.execute_input":"2024-10-22T03:53:02.964905Z","iopub.status.idle":"2024-10-22T03:53:03.008871Z","shell.execute_reply.started":"2024-10-22T03:53:02.964847Z","shell.execute_reply":"2024-10-22T03:53:03.007106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sii\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:53:13.085896Z","iopub.execute_input":"2024-10-22T03:53:13.086450Z","iopub.status.idle":"2024-10-22T03:53:13.102136Z","shell.execute_reply.started":"2024-10-22T03:53:13.086400Z","shell.execute_reply":"2024-10-22T03:53:13.100402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"markdown","source":"Detecting missing data visually using Missingno library","metadata":{}},{"cell_type":"code","source":"import missingno as msno\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:56:37.403397Z","iopub.execute_input":"2024-10-22T03:56:37.404541Z","iopub.status.idle":"2024-10-22T03:56:37.428302Z","shell.execute_reply.started":"2024-10-22T03:56:37.404487Z","shell.execute_reply":"2024-10-22T03:56:37.426952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.bar(train.iloc[:, :82], sort='ascending') ","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:59:36.677287Z","iopub.execute_input":"2024-10-22T03:59:36.677747Z","iopub.status.idle":"2024-10-22T03:59:40.105616Z","shell.execute_reply.started":"2024-10-22T03:59:36.677706Z","shell.execute_reply":"2024-10-22T03:59:40.103654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the dendrogram\ntry:\n    # Create the dendrogram\n    msno.dendrogram(train)\nexcept ValueError:\n    # Ignore the ValueError\n    pass","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:56:58.639794Z","iopub.execute_input":"2024-10-22T03:56:58.640226Z","iopub.status.idle":"2024-10-22T03:57:00.141564Z","shell.execute_reply.started":"2024-10-22T03:56:58.640184Z","shell.execute_reply":"2024-10-22T03:57:00.140514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.heatmap(train.iloc[:,:])","metadata":{"execution":{"iopub.status.busy":"2024-10-22T03:58:22.811659Z","iopub.execute_input":"2024-10-22T03:58:22.812106Z","iopub.status.idle":"2024-10-22T03:58:33.573968Z","shell.execute_reply.started":"2024-10-22T03:58:22.812062Z","shell.execute_reply":"2024-10-22T03:58:33.572461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['basic_demos_enroll_season', 'cgas_season', 'physical_season', \n                       'fgc_season', 'bia_season', 'pciat_season', 'sds_season', 'preint_eduhx_season']\n\nplt.figure(figsize=(18, 20))\nsns.set(style=\"whitegrid\")\n\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(4, 2, i)  # Create subplots: 4 rows, 2 columns, plot index i\n    sns.boxplot(x=col, y='sii', data=train, palette=\"Set2\")\n    \n\n    plt.xlabel(f'{col}', fontsize=12)\n    plt.ylabel('Severity Impairment Index (sii)', fontsize=12)\n    plt.title(f\"Distribution of 'sii' by {col}\", fontsize=14)\n    \n    plt.xticks(rotation=30 if train[col].nunique() > 5 else 0)\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-22T04:06:12.830092Z","iopub.execute_input":"2024-10-22T04:06:12.830611Z","iopub.status.idle":"2024-10-22T04:06:16.509318Z","shell.execute_reply.started":"2024-10-22T04:06:12.830565Z","shell.execute_reply":"2024-10-22T04:06:16.507214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols = train.select_dtypes(include=['float64', 'int64']).columns\nplots_per_row = 4  \nn_rows = (len(numerical_cols) + plots_per_row - 1) // plots_per_row\n\nplt.figure(figsize=(22, 5 * n_rows))\nsns.set(style=\"whitegrid\")\n\nfor i, col in enumerate(numerical_cols):\n    plt.subplot(n_rows, plots_per_row, i + 1)\n    sns.boxplot(x='sii', y=col, data=train, palette=\"Set3\", showfliers=False)\n    sns.stripplot(x='sii', y=col, data=train, color='black', size=3, alpha=0.5, jitter=True)\n    \n    plt.title(f\"'{col}' vs 'sii'\", fontsize=13)\n    plt.xlabel('Severity Impairment Index (sii)', fontsize=12)\n    plt.ylabel(col, fontsize=12)\n    \n    plt.tight_layout(pad=1.0)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T04:02:59.903597Z","iopub.execute_input":"2024-10-22T04:02:59.904137Z","iopub.status.idle":"2024-10-22T04:05:10.962184Z","shell.execute_reply.started":"2024-10-22T04:02:59.904078Z","shell.execute_reply":"2024-10-22T04:05:10.960654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 20))\n\n# Create subplots for each numerical column\nfor i, col in enumerate(numerical_cols, 1):\n    plt.subplot((len(numerical_cols) + plots_per_row - 1) // plots_per_row, plots_per_row, i)  # Create subplots\n    sns.histplot(train[col], kde=True, bins=30, color='blue')\n    plt.xlabel(col, fontsize=12)\n    plt.ylabel('Frequency', fontsize=12)\n    plt.title(f'Distribution of {col}', fontsize=14)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-22T04:06:39.742856Z","iopub.execute_input":"2024-10-22T04:06:39.743399Z","iopub.status.idle":"2024-10-22T04:07:12.515703Z","shell.execute_reply.started":"2024-10-22T04:06:39.743322Z","shell.execute_reply":"2024-10-22T04:07:12.514272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"markdown","source":"## Correlation Matrix","metadata":{}},{"cell_type":"code","source":"season_cols = ['basic_demos_enroll_season', \n                'cgas_season', \n                'physical_season', \n                'fgc_season', \n                'bia_season', \n                'pciat_season', \n                'sds_season', \n                'preint_eduhx_season',\n                'paq_a_season',\n                'paq_c_season',\n                'fitness_endurance_season'\n]\nseason_mapping = {\n    'Spring': 0,\n    'Summer': 1,\n    'Fall': 2,\n    'Winter': 3\n}\nfor col in season_cols:\n    if col in train.columns:\n        train[col] = train[col].replace(season_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-10-22T04:14:17.696606Z","iopub.execute_input":"2024-10-22T04:14:17.697060Z","iopub.status.idle":"2024-10-22T04:14:17.711809Z","shell.execute_reply.started":"2024-10-22T04:14:17.697017Z","shell.execute_reply":"2024-10-22T04:14:17.709942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_no_id = train.drop(columns=['id'], errors='ignore')\ncorrelation_matrix = train_data_no_id.corr()\nplt.figure(figsize=(30, 30))\nsns.heatmap(correlation_matrix, annot=True, fmt='.1f', cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-22T04:14:21.755667Z","iopub.execute_input":"2024-10-22T04:14:21.756108Z","iopub.status.idle":"2024-10-22T04:14:39.418365Z","shell.execute_reply.started":"2024-10-22T04:14:21.756065Z","shell.execute_reply":"2024-10-22T04:14:39.417107Z"},"trusted":true},"execution_count":null,"outputs":[]}]}