{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Introduction\nFirst day EDA - I am interested in what makes a 'severe' addicted distribution and how they compare to level 0 (least 'severe') addicted distribution. Let us draw some plots.\n","metadata":{}},{"cell_type":"markdown","source":"Quick highlights\n\n1. Younger kids less than 10 years old have never been classified as level 3.0 severe (max severity). This is either bias from examiners or they are incapable of reaching this level of severity. Or just luck.\n2. There are more males than females in the data. At first glance, the percentage of 3.0 severe vs 0.0 severe doesn't seem too different between genders.\n3. Body weight, height are highly correlated to age so they show similar trends in distribution.\n4. Clear differences in physical endurance level between 3.0 severity and 0.0 severity.\n5. Clear shift in distribution in columns such as PreInt_EduHx_computerinternet_hoursday, SDS_SDS_total_raw, but I'm not sure what these are yet. TODO\n\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-22T00:04:14.007658Z","iopub.execute_input":"2024-09-22T00:04:14.008121Z","iopub.status.idle":"2024-09-22T00:04:14.833812Z","shell.execute_reply.started":"2024-09-22T00:04:14.008079Z","shell.execute_reply":"2024-09-22T00:04:14.832714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-21T23:51:23.691195Z","iopub.execute_input":"2024-09-21T23:51:23.691675Z","iopub.status.idle":"2024-09-21T23:51:23.801350Z","shell.execute_reply.started":"2024-09-21T23:51:23.691631Z","shell.execute_reply":"2024-09-21T23:51:23.799702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.sii.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-09-21T23:52:13.379271Z","iopub.execute_input":"2024-09-21T23:52:13.379721Z","iopub.status.idle":"2024-09-21T23:52:13.393089Z","shell.execute_reply.started":"2024-09-21T23:52:13.379670Z","shell.execute_reply":"2024-09-21T23:52:13.391680Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_severe = df_train[df_train['sii']==3.0]\ndf_norm = df_train[df_train['sii']==0.0]","metadata":{"execution":{"iopub.status.busy":"2024-09-22T00:00:19.701048Z","iopub.execute_input":"2024-09-22T00:00:19.701507Z","iopub.status.idle":"2024-09-22T00:00:19.710565Z","shell.execute_reply.started":"2024-09-22T00:00:19.701448Z","shell.execute_reply":"2024-09-22T00:00:19.709331Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Following function visualize 3.0 severe vs 0.0 severe on all columns\n# It automatically chooses which plot is best to draw, it could be improved (for example, it treats some categorical columns as continuous) TODO\ndef analyze_and_visualize_dataframe(df, severity_col, severe_value=3.0, normal_value=0.0):\n    # Split the dataframe\n    df_severe = df[df[severity_col] == severe_value]\n    df_norm = df[df[severity_col] == normal_value]\n    \n    # Get list of columns (excluding the severity column and ID-like columns)\n    columns = [col for col in df.columns if col != severity_col and df[col].nunique() < len(df) * 0.5]\n    \n    for col in columns:\n        plt.figure(figsize=(12, 5))\n        \n        # Determine the type of data in the column\n        if df[col].dtype in ['int64', 'float64'] and df[col].nunique() > 2:\n            # Numerical data (non-binary): Use histogram and box plot\n            plt.subplot(1, 2, 1)\n            sns.histplot(data=df_severe, x=col, kde=True, color='red', label='Severe (3)')\n            sns.histplot(data=df_norm, x=col, kde=True, color='blue', label='Normal (0)', alpha=0.5)\n            plt.title(f'Distribution of {col}')\n            plt.legend()\n            \n            plt.subplot(1, 2, 2)\n            sns.boxplot(data=df, x=severity_col, y=col, order=[normal_value, severe_value])\n            plt.title(f'Box Plot of {col}')\n            plt.xticks([0, 1], ['Normal (0)', 'Severe (3)'])\n            \n        else:\n            # Categorical data (including binary): Use bar plot\n            plt.subplot(1, 2, 1)\n            df_severe[col].value_counts(normalize=True).plot(kind='bar', alpha=0.7, color='red')\n            plt.title(f'Distribution of {col} (Severe)')\n            plt.ylabel('Proportion')\n            \n            plt.subplot(1, 2, 2)\n            df_norm[col].value_counts(normalize=True).plot(kind='bar', alpha=0.7, color='blue')\n            plt.title(f'Distribution of {col} (Normal)')\n            plt.ylabel('Proportion')\n        \n        plt.tight_layout()\n        plt.show()\n        \n        # For binary columns, add a comparison plot\n        if df[col].nunique() == 2:\n            plt.figure(figsize=(8, 6))\n            sns.countplot(data=df, x=col, hue=severity_col, hue_order=[normal_value, severe_value])\n            plt.title(f'Comparison of {col} Distribution')\n            plt.xlabel(col)\n            plt.ylabel('Count')\n            plt.legend(title='Severity', labels=['Normal (0)', 'Severe (3)'])\n            plt.tight_layout()\n            plt.show()\n    \n    # Create correlation heatmaps\n    numeric_columns = df.select_dtypes(include=['int64', 'float64']).columns\n    plt.figure(figsize=(20, 8))\n    \n    plt.subplot(1, 2, 1)\n    sns.heatmap(df_severe[numeric_columns].corr(), annot=False, cmap='coolwarm')\n    plt.title('Correlation Heatmap (Severe)')\n    \n    plt.subplot(1, 2, 2)\n    sns.heatmap(df_norm[numeric_columns].corr(), annot=False, cmap='coolwarm')\n    plt.title('Correlation Heatmap (Normal)')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T00:15:20.776330Z","iopub.execute_input":"2024-09-22T00:15:20.777444Z","iopub.status.idle":"2024-09-22T00:15:20.795541Z","shell.execute_reply.started":"2024-09-22T00:15:20.777393Z","shell.execute_reply":"2024-09-22T00:15:20.794291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"analyze_and_visualize_dataframe(df_train, 'sii', 3, 0)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T00:15:21.012279Z","iopub.execute_input":"2024-09-22T00:15:21.012718Z","iopub.status.idle":"2024-09-22T00:16:15.970246Z","shell.execute_reply.started":"2024-09-22T00:15:21.012666Z","shell.execute_reply":"2024-09-22T00:16:15.968972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next step\n- I want to EDA focusing on one specific 3.0 severe child and try to tell a story about him/her. Or maybe I'll do one female and one male.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}