{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":206675453,"sourceType":"kernelVersion"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[{"file_id":"1P9x2gc8VzxTgj7Jwd_XXSmRl4vxnuPI7","timestamp":1733668340594},{"file_id":"https://storage.googleapis.com/kaggle-colab-exported-notebooks/hms-data-exploration-ce7f528a-9f71-468e-8ad9-0e214724babe.ipynb?X-Goog-Algorithm=GOOG4-RSA-SHA256&X-Goog-Credential=gcp-kaggle-com%40kaggle-161607.iam.gserviceaccount.com/20241202/auto/storage/goog4_request&X-Goog-Date=20241202T141659Z&X-Goog-Expires=259200&X-Goog-SignedHeaders=host&X-Goog-Signature=197f2bfa2cd59d0f362f6533df00995eacc39535ba5b8f2c41a0c6c33e5f10e170feb7efbfce344ec652b038d007b9cbd4703d41b4aa4b1d934207871a41337161dcf57b2820885e61b851197a2209127cbe81fb7b73139a45151da83e3b27327e15a548d36f1bf6b84006b5541061b88acb387988a02aa964d4f24cf7ebc61d1ceca3e1a35c1564178518bf147736f88038f9b23e22e920f824604fbc51edf610ba362c2fe2a88b141775b5423514626721a4302eb028d7fbd25cbee2cdfb20a7797ee97e0e53ff775badbaab09aa010dd409bb118a9ec1f0f236787370c8fa3d151fd02ae3774c3a2b340b8fd788124a631c942321ae0cb1d92d1675f17cd3","timestamp":1733149320630}],"toc_visible":true,"collapsed_sections":["AbiCDsUup8Us"]},"widgets":{"application/vnd.jupyter.widget-state+json":{"955fdd1af1df441aacc1e8d397f3a6d3":{"model_module":"@jupyter-widgets/controls","model_name":"VBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"VBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"VBoxView","box_style":"","children":["IPY_MODEL_b170f3acabef43fbbfeab509b71d5ef6","IPY_MODEL_29d68a6b1e974f3abfa8971d334536b0","IPY_MODEL_a4ffa657117244df8e871877eedb1e61","IPY_MODEL_454f4abf585f4122a63adad934b19133","IPY_MODEL_a564ad2e59c547a881dddf138ec8491b"],"layout":"IPY_MODEL_3b22a52c71b44c93a71ae3db60047bc3"}},"b170f3acabef43fbbfeab509b71d5ef6":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_0f71d72bd6c249ec97be506a240acdee","placeholder":"​","style":"IPY_MODEL_78d0c4395b4748bc9b7dbc0527300136","value":"<center> <img\nsrc=https://www.kaggle.com/static/images/site-logo.png\nalt='Kaggle'> <br> Create an API token from <a\nhref=\"https://www.kaggle.com/settings/account\" target=\"_blank\">your Kaggle\nsettings page</a> and paste it below along with your Kaggle username. <br> </center>"}},"29d68a6b1e974f3abfa8971d334536b0":{"model_module":"@jupyter-widgets/controls","model_name":"TextModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"TextModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"TextView","continuous_update":true,"description":"Username:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_64b90cfca4d040d19e94a62c60cbf499","placeholder":"​","style":"IPY_MODEL_142e108187ef4c62b3ae53325bc251cc","value":""}},"a4ffa657117244df8e871877eedb1e61":{"model_module":"@jupyter-widgets/controls","model_name":"PasswordModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"PasswordModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"PasswordView","continuous_update":true,"description":"Token:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_9ba4765ee4f8439fabb63d42186b3fff","placeholder":"​","style":"IPY_MODEL_c3f69200e85f43d39702a3cb511cf63a","value":""}},"454f4abf585f4122a63adad934b19133":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ButtonView","button_style":"","description":"Login","disabled":false,"icon":"","layout":"IPY_MODEL_9aaabfb007a7407f875b4aa66e73801e","style":"IPY_MODEL_29cc344ff5ec4e33b6e5715b82a1f6d9","tooltip":""}},"a564ad2e59c547a881dddf138ec8491b":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_4bab31b4a89540f0864b2fe820fe8619","placeholder":"​","style":"IPY_MODEL_a71688ce002b4190b04ee995c6fb7b2e","value":"\n<b>Thank You</b></center>"}},"3b22a52c71b44c93a71ae3db60047bc3":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":"center","align_self":null,"border":null,"bottom":null,"display":"flex","flex":null,"flex_flow":"column","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"50%"}},"0f71d72bd6c249ec97be506a240acdee":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"78d0c4395b4748bc9b7dbc0527300136":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"64b90cfca4d040d19e94a62c60cbf499":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"142e108187ef4c62b3ae53325bc251cc":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"9ba4765ee4f8439fabb63d42186b3fff":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c3f69200e85f43d39702a3cb511cf63a":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"9aaabfb007a7407f875b4aa66e73801e":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"29cc344ff5ec4e33b6e5715b82a1f6d9":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","button_color":null,"font_weight":""}},"4bab31b4a89540f0864b2fe820fe8619":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a71688ce002b4190b04ee995c6fb7b2e":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# IMPORTANT: SOME KAGGLE DATA SOURCES ARE PRIVATE\n\n# RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES.\n\nimport kagglehub\n\nkagglehub.login()\n","metadata":{"id":"qGP-hQzbu2UM","executionInfo":{"status":"ok","timestamp":1733659663786,"user_tz":-120,"elapsed":284,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"106b32b8-206d-47a0-b588-fe6c15e79440","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# IMPORTANT: RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES,\n\n# THEN FEEL FREE TO DELETE THIS CELL.\n\n# NOTE: THIS NOTEBOOK ENVIRONMENT DIFFERS FROM KAGGLE'S PYTHON\n\n# ENVIRONMENT SO THERE MAY BE MISSING LIBRARIES USED BY YOUR\n\n# NOTEBOOK.\n\n\n\nchild_mind_institute_problematic_internet_use_path = kagglehub.competition_download('child-mind-institute-problematic-internet-use')\n\n\n\nprint('Data source import complete.')\n","metadata":{"id":"h_nz4Mtou2UN","executionInfo":{"status":"ok","timestamp":1733659665710,"user_tz":-120,"elapsed":1275,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"d1bf325c-e7b1-4afa-a1ab-cdde15cd61bd","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**This notebook is focusing on HMS EDA**","metadata":{"id":"XDHIFr5-u2UN"}},{"cell_type":"code","source":"palette='viridis'","metadata":{"id":"-zuJL2ucPFgb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Libraries","metadata":{"id":"D14pZZaVu2UO"}},{"cell_type":"code","source":"import pandas as pd\n\nfrom sklearn.base import BaseEstimator, TransformerMixin\n\nimport numpy as np\n\npd.set_option('display.max_columns', 500)\n\npd.set_option('display.max_rows', 100)\n\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns","metadata":{"trusted":true,"id":"N_lCHM6Xu2UO"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper Functions","metadata":{"id":"qvqHEsofVM4K"}},{"cell_type":"code","source":"def filter_by_instrument(df_train, df_dict, instrument_filter):\n\n    \"\"\"\n\n    Filter the training dataset by a specific instrument using the dictionary file.\n\n\n\n    Parameters:\n\n        df_train (pd.DataFrame): The training dataset.\n\n        df_dict (pd.DataFrame): The dictionary file.\n\n        instrument_filter (str): Instrument name to filter columns.\n\n\n\n    Returns:\n\n        pd.DataFrame: Filtered training data for the specified instrument.\n\n        pd.DataFrame: Filtered dictionary for the specified instrument.\n\n    \"\"\"\n\n    df_dict_instrument = df_dict[df_dict['Instrument'] == instrument_filter]\n\n    columns = df_dict_instrument['Field'].tolist()\n\n    df_filtered = df_train[columns]\n\n    return df_filtered, df_dict_instrument\n\n\n\n\n\ndef general_info(df, name=\"Dataset\"):\n\n    \"\"\"\n\n    Display general information about the dataset.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset to analyze.\n\n        name (str): The name of the dataset for display purposes.\n\n    \"\"\"\n\n    print(f\"Summary of {name}:\")\n\n    print(df.info())\n\n    print(\"\\nSummary Statistics:\")\n\n    print(df.describe())\n\n    print(\"\\nMissing Values Percentage:\")\n\n    print(df.isnull().sum() / len(df))\n\n    total_rows = df.shape[0]\n\n\n\n    completely_missing_rows = df[df.isnull().all(axis=1)].shape[0]\n\n    # Total missing values\n\n    total_missing_values = df.isnull().sum().sum()\n\n    percentage_completely_missing_rows = (completely_missing_rows / total_rows) * 100\n\n    print(\"\\n Totally Missing Rows Percentage:\")\n\n    print(f\"{percentage_completely_missing_rows:.2f}\")\n\n\n\ndef analyze_categorical(df, categorical_columns):\n\n    \"\"\"\n\n    Analyze categorical columns by printing value counts.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        categorical_columns (list): List of categorical column names.\n\n    \"\"\"\n\n    for col in categorical_columns:\n\n        print(f\"\\nValue Counts for {col}:\")\n\n        print(df[col].value_counts())\n\n\n\ndef analyze_numerical(df, numerical_columns):\n\n    \"\"\"\n\n    Analyze numerical columns by generating descriptive statistics.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        numerical_columns (list): List of numerical column names.\n\n    \"\"\"\n\n    print(\"\\nDescriptive Statistics for Numerical Columns:\")\n\n    print(df[numerical_columns].describe())\n\n\n\ndef plot_numerical_distributions(df, numerical_columns):\n\n    \"\"\"\n\n    Plot distributions for numerical columns.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        numerical_columns (list): List of numerical column names.\n\n    \"\"\"\n\n    for col in numerical_columns:\n\n        plt.figure(figsize=(8, 4))\n\n        sns.histplot(df[col], kde=True, bins=30)\n\n        plt.title(f'Distribution of {col}')\n\n        plt.show()\n\ndef plot_categorical_distributions(df, categorical_columns):\n\n    \"\"\"\n\n    Plot bar charts for categorical columns.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        categorical_columns (list): List of categorical column names.\n\n    \"\"\"\n\n    for col in categorical_columns:\n\n        plt.figure(figsize=(8, 4))\n\n        df[col].value_counts().plot(kind='bar', color='skyblue')\n\n        plt.title(f'Distribution of {col}')\n\n        plt.ylabel('Count')\n\n        plt.xlabel('Categories')\n\n        plt.xticks(rotation=45)\n\n        plt.show()","metadata":{"id":"tSwqg1zXVOcm","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define Directories","metadata":{"id":"KSYcSBhfu2UP"}},{"cell_type":"code","source":"root_path = child_mind_institute_problematic_internet_use_path\n\ncsv_train_path = f'{root_path}/train.csv'\n\ncsv_dict_path = f'{root_path}/data_dictionary.csv'\n\n\n\ncsv_test_path = f'{root_path}/test.csv'","metadata":{"trusted":true,"id":"Gd41U2K5u2UP"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load CSV Data","metadata":{"id":"ITnsXS3uu2UP"}},{"cell_type":"code","source":"df_train_csv = pd.read_csv(csv_train_path)\n\ndf_test_csv = pd.read_csv(csv_test_path)\n\ndf_dict_csv = pd.read_csv(csv_dict_path)\n\nunique_instruments = np.unique(df_dict_csv['Instrument'])\n\nprint(f\"Unique Instruments are\\n  {unique_instruments}\")\n\nprint(f\"The number of Unique Instruments is  {len(unique_instruments)}\")","metadata":{"trusted":true,"id":"uVZRGgtBu2UP","executionInfo":{"status":"ok","timestamp":1733664356948,"user_tz":-120,"elapsed":6,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"66584416-e54d-4105-e7e6-9af7da1fc095","collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_csv.columns","metadata":{"id":"uRapE0sSV4pq","executionInfo":{"status":"ok","timestamp":1733664357377,"user_tz":-120,"elapsed":434,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"a012e240-42bd-4a1b-bada-d697148bc9eb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_columns = df_train_csv.columns\n\ntest_csv_columns = df_test_csv.columns","metadata":{"trusted":true,"id":"QgPL0swCu2UQ"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of training columns is: \", len(train_csv_columns))\n\nprint(\"Number of test columns is: \", len(test_csv_columns))","metadata":{"id":"9SyZ8HaRntMU","executionInfo":{"status":"ok","timestamp":1733664357377,"user_tz":-120,"elapsed":5,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"d9783b3e-2b90-4f52-fcfd-a0ccc3edf9a9","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# different columns\n\ndiff_cols = set(train_csv_columns) - set(test_csv_columns)\n\ndiff_cols\n\ndiff_cols.remove('sii')","metadata":{"id":"nlcFkUcTn3DK","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of PCIAT columns\n\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\n\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv[PCIAT_cols].sum(axis=1)\n\n# convert 0 to NaN\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv['Calculated_PCIAT_Total'].replace(0, np.nan)\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv['Calculated_PCIAT_Total'].replace(0.0, np.nan)\n\ndf_train_csv['PCIAT-PCIAT_Total'] = df_train_csv['PCIAT-PCIAT_Total'].replace(0.0, np.nan)\n\n\n\n# Count mismatched totals\n\nmiscalculated_PCIAT_Total = (df_train_csv['PCIAT-PCIAT_Total'] != df_train_csv['Calculated_PCIAT_Total'])\n\n\n\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\n\n\ndef recalculate_sii(row):\n\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n\n        return np.nan\n\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n\n        return 0\n\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n\n        return 1\n\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n\n        return 2\n\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n\n        return 3\n\n    return np.nan\n\ndf_train_csv['Severity Impairment Index (SII)'] = df_train_csv.apply(recalculate_sii, axis=1)\n","metadata":{"id":"OlGgXBcUH_Bt","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Define the mapping using pd.cut\n\nbins = [5 ,14, 22]\n\nlabels = ['Child', 'Teen']\n\n\n\ndf_train_csv['Age Group'] = pd.cut(df_train_csv['Basic_Demos-Age'], bins=bins, labels=labels, right=True, include_lowest=True)\n","metadata":{"id":"y7roB_CXM8NZ","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_csv","metadata":{"id":"wzIXkb64cNKj","executionInfo":{"status":"ok","timestamp":1733664362330,"user_tz":-120,"elapsed":5,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"61dfa6ee-7c54-477f-8dbe-17ad612c8993","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Map to meaningful names","metadata":{"id":"pia5T1cvafPm"}},{"cell_type":"code","source":"# Define a clean, readable mapping for PCIAT fields\n\npciat_mapping = {\n\n    'PCIAT-PCIAT_01': 'Disobeys Time Limits',\n\n    'PCIAT-PCIAT_02': 'Neglects Chores',\n\n    'PCIAT-PCIAT_03': 'Prefers Online Over Family',\n\n    'PCIAT-PCIAT_04': 'Forms Online Relationships',\n\n    'PCIAT-PCIAT_05': 'Complaints About Time Online',\n\n    'PCIAT-PCIAT_06': 'Grades Suffer',\n\n    'PCIAT-PCIAT_07': 'Checks Email First',\n\n    'PCIAT-PCIAT_08': 'Withdrawn From Others',\n\n    'PCIAT-PCIAT_09': 'Defensive About Online Use',\n\n    'PCIAT-PCIAT_10': 'Caught Sneaking Online',\n\n    'PCIAT-PCIAT_11': 'Alone Time on Computer',\n\n    'PCIAT-PCIAT_12': 'Receives Strange Calls',\n\n    'PCIAT-PCIAT_13': 'Snaps When Interrupted',\n\n    'PCIAT-PCIAT_14': 'More Fatigued Than Before',\n\n    'PCIAT-PCIAT_15': 'Preoccupied With Being Online',\n\n    'PCIAT-PCIAT_16': 'Throws Tantrums Over Limits',\n\n    'PCIAT-PCIAT_17': 'Quits Hobbies for Online',\n\n    'PCIAT-PCIAT_18': 'Angry About Time Limits',\n\n    'PCIAT-PCIAT_19': 'Prefers Online Over Friends',\n\n    'PCIAT-PCIAT_20': 'Depressed When Offline',\n\n}\n\n\n\n# Apply the mapping to rename the columns\n\ndf_train_csv = df_train_csv.rename(columns=pciat_mapping)\n\n\n\n\n\n# create dictionary\n\nfields_names_map = dict(zip(df_dict_csv['Field'], df_dict_csv['Description']))\n\n# remove all PCIAT columns\n\nfields_names_map = {k: v for k, v in fields_names_map.items() if 'PCIAT' not in k}\n\ndf_train_csv = df_train_csv.rename(columns=fields_names_map)\n\ndf_test_csv = df_test_csv.rename(columns=fields_names_map)\n\nquestion_columns = list(pciat_mapping.values())","metadata":{"id":"Eos058SxIB29","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sii_labels_mapping = {\n\n    '0.0': 'None',\n\n    '1.0': 'Mild',\n\n    '2.0': 'Moderate',\n\n    '3.0': 'Severe'\n\n}\n\n# Define a mapping for Internet usage hours\n\ninternet_hours_mapping = {\n\n    0.0: '0-1 Hours',\n\n    1.0: '1-2 Hours',\n\n    2.0: '2-3 Hours',\n\n    3.0: '3+ Hours'\n\n}","metadata":{"id":"hWm2AUbOfa_O","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Drop Seasons","metadata":{"id":"BCn8LVqJbyCQ"}},{"cell_type":"code","source":"# drop all columns that has word season\n\ndf_train_csv = df_train_csv.drop(columns=[col for col in df_train_csv.columns if 'Season' in col])\n\ndf_test_csv = df_test_csv.drop(columns=[col for col in df_test_csv.columns if 'Season' in col])","metadata":{"id":"ivRzZfgqbN9B","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fixing Labels","metadata":{"id":"AbiCDsUup8Us"}},{"cell_type":"markdown","source":"We have a lot of missing values in the target","metadata":{"id":"Y1lxopj6p--s"}},{"cell_type":"code","source":"# visualize how many incorrect","metadata":{"id":"Z7RT2ycYFP-0","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of PCIAT columns\n\n\n\n# Multiply relevant columns by 5 and calculate the sum across the row\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv[question_columns].sum(axis=1)\n\n# convert 0 to NaN\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv['Calculated_PCIAT_Total'].replace(0, np.nan)\n\ndf_train_csv['Calculated_PCIAT_Total'] = df_train_csv['Calculated_PCIAT_Total'].replace(0.0, np.nan)\n\ndf_train_csv['PCIAT-PCIAT_Total'] = df_train_csv['PCIAT-PCIAT_Total'].replace(0.0, np.nan)\n\n\n\n# Count mismatched totals\n\nmiscalculated_PCIAT_Total = (df_train_csv['PCIAT-PCIAT_Total'] != df_train_csv['Calculated_PCIAT_Total']).sum()\n\nprint(f\"Number of Miscalculated PCIAT Totals: {miscalculated_PCIAT_Total}\")\n\n\n","metadata":{"collapsed":true,"id":"jwcM_UiPtTA9","executionInfo":{"status":"ok","timestamp":1733664369921,"user_tz":-120,"elapsed":3,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"fdbb78b6-401d-494c-cc63-8144dd2b4cde","jupyter":{"outputs_hidden":true},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"differences = df_train_csv[\n\n    ~((df_train_csv['Calculated_PCIAT_Total'] == df_train_csv['PCIAT-PCIAT_Total']) |\n\n      (df_train_csv['Calculated_PCIAT_Total'].isna() & df_train_csv['PCIAT-PCIAT_Total'].isna()))\n\n]\n\n(differences[['PCIAT-PCIAT_Total', 'Calculated_PCIAT_Total']])\n","metadata":{"id":"2SALWAjOvZSs","executionInfo":{"status":"ok","timestamp":1733664371057,"user_tz":-120,"elapsed":5,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"f527dddf-53a5-4c92-9f8e-24241709a5c0","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_csv.iloc[1,:]\n\ndf_train_csv[(df_train_csv['sii']==0.0) & (df_train_csv['Severity Impairment Index (SII)'].isna()) ]","metadata":{"id":"WW1fvbX-xvM8","executionInfo":{"status":"ok","timestamp":1733664374884,"user_tz":-120,"elapsed":349,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"bfa7d710-3c54-451a-c4b3-e8280fc9ad4c","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visuals","metadata":{"id":"67XMxMGHq00a"}},{"cell_type":"markdown","source":"## Missing Labels before and after","metadata":{"id":"vi6M1XPcgHbk"}},{"cell_type":"code","source":"\n\n# Prepare data for plotting\n\ncounts_before = df_train_csv['sii'].isna().value_counts(normalize=True) * 100\n\ncounts_after = df_train_csv['Severity Impairment Index (SII)'].isna().value_counts(normalize=True) * 100\n\n\n\n# Ensure both categories are represented\n\ncounts_before = counts_before.reindex([False, True], fill_value=0)\n\ncounts_after = counts_after.reindex([False, True], fill_value=0)\n\n\n\n# Set up the plot style for professional presentation\n\nsns.set_context(\"poster\")\n\n\n\n# Seaborn barplot version\n\nplt.figure(figsize=(16, 10))\n\n\n\n# Prepare data for plotting\n\nplot_data = pd.DataFrame({\n\n    'Missing Status': ['Non-Missing', 'Missing'],\n\n    'Before': [counts_before[False], counts_before[True]],\n\n    'After': [counts_after[False], counts_after[True]]\n\n})\n\n\n\n# Melt the DataFrame for easier plotting\n\nplot_data_melted = plot_data.melt(\n\n    id_vars='Missing Status',\n\n    var_name='Stage',\n\n    value_name='Percentage'\n\n)\n\n\n\n# Create the bar plot\n\nax = sns.barplot(\n\n    x='Missing Status',\n\n    y='Percentage',\n\n    hue='Stage',\n\n    data=plot_data_melted,\n\n    palette='viridis'\n\n)\n\n\n\n# Customize the plot\n\nplt.title('Missing Labels Comparison Before and After Fixing', fontsize=24, fontweight='bold')\n\nplt.xlabel('Labels Status', fontsize=20)\n\nplt.ylabel('Percentage (%)', fontsize=20)\n\nplt.xticks(fontsize=16)\n\nplt.yticks(fontsize=16)\n\n\n\n# Add percentage annotations\n\n# Add percentage annotations with correct alignment\n\nbar_width = 0.4  # Matches bar width in the grouped plot\n\nfor i, status in enumerate(['Non-Missing', 'Missing']):\n\n    status_data = plot_data_melted[plot_data_melted['Missing Status'] == status]\n\n    for j, row in enumerate(status_data.itertuples()):\n\n        # Calculate position for each group\n\n        x_pos = i - bar_width / 2 + j * bar_width  # Adjust for grouped bars\n\n        plt.text(\n\n            x_pos,\n\n            row.Percentage + 1,  # Place the text slightly above the bar\n\n            f'{row.Percentage:.1f}%',\n\n            ha='center',\n\n            va='bottom',\n\n            fontsize=12,\n\n            fontweight='bold'\n\n        )\n\n\n\nplt.legend(title='Fixing Stage', title_fontsize=16, fontsize=14, loc='upper right')\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"EfXQRqV1HykL","executionInfo":{"status":"ok","timestamp":1733664383911,"user_tz":-120,"elapsed":1270,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"9bbecaad-5a4e-4964-df55-5a83a513d042","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DROP INCORRECT SII","metadata":{"id":"bk50uURpczI6"}},{"cell_type":"code","source":"df_train_csv = df_train_csv.dropna(subset=['sii'])","metadata":{"id":"IWxkj-sHc1nt","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rename columns for clarity in the heatmap\n\ndf_corr = df_train_csv.rename(\n\n    columns={\n\n        'Age of participant': 'Age',\n\n        'Hours of using computer/internet': 'Internet Hours/Day',\n\n    }\n\n)\n\ndf_corr['Internet Hours/Day']","metadata":{"id":"1C8FQvT8bkZB","executionInfo":{"status":"ok","timestamp":1733664457016,"user_tz":-120,"elapsed":230,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"19441b6c-263b-4b81-8a6c-c94eb19abe31","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation Between Age, Internet Usage, and SII","metadata":{"id":"Cz9DZLTHgKqE"}},{"cell_type":"code","source":"# Rename columns for clarity in the heatmap\n\ndf_corr = df_train_csv.rename(\n\n    columns={\n\n        'Age of participant': 'Age',\n\n        'Hours of using computer/internet': 'Internet Hours/Day',\n\n    }\n\n)\n\n# Calculate the correlation matrix\n\ncorr_data = df_corr[['Age', 'Internet Hours/Day', 'Severity Impairment Index (SII)']].corr()\n\n\n\n# Create a heatmap with professional styling\n\nplt.figure(figsize=(16, 10))\n\nsns.heatmap(\n\n    corr_data,\n\n    annot=True,\n\n    cmap='coolwarm',\n\n    fmt='.2f',\n\n    linewidths=1.5,\n\n    linecolor='black',\n\n    cbar_kws={'shrink': 0.8, 'label': 'Correlation Coefficient'},\n\n    annot_kws={'fontsize': 14, 'fontweight': 'bold'}\n\n)\n\n\n\n# Add titles and labels\n\nplt.title('Correlation Between Age, Internet Usage, and SII', fontsize=20, fontweight='bold')\n\nplt.xticks(fontsize=14, rotation=45, ha='right', fontweight='bold')\n\nplt.yticks(fontsize=14, fontweight='bold')\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"6yWNmHm0XNVv","executionInfo":{"status":"ok","timestamp":1733664462817,"user_tz":-120,"elapsed":1219,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"9e940534-17d5-4fc5-b8b1-d1d781ca9d3e","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Distribution of SII Levels by Age Group and Internet Usage Hours","metadata":{"id":"YmeSLfemgA23"}},{"cell_type":"code","source":"\n\n# Group data and calculate the distribution of SII\n\npivot_data = df_train_csv.groupby(\n\n    ['Age Group', 'Hours of using computer/internet', 'Severity Impairment Index (SII)']\n\n).size().reset_index(name='Count')\n\n\n\n# Replace the 'Hours of using computer/internet' with descriptive labels\n\npivot_data['Hours of using computer/internet'] = pivot_data['Hours of using computer/internet'].map(internet_hours_mapping)\n\n\n\n# Pivot table\n\npivot_table = pivot_data.pivot_table(\n\n    index=['Age Group', 'Hours of using computer/internet'],\n\n    columns='Severity Impairment Index (SII)',\n\n    values='Count',\n\n    fill_value=0\n\n)\n\n\n\n# Normalize the data to proportions\n\npivot_normalized = pivot_table.div(pivot_table.sum(axis=1), axis=0)\n\n\n\n# Sort rows explicitly by SII = 3.0 values\n\nif 3.0 in pivot_normalized.columns:\n\n    pivot_normalized = pivot_normalized.sort_values(by=3.0, ascending=False)\n\n\n\n# Plot the Stacked Bar Plot\n\nfig, ax = plt.subplots(figsize=(14, 8))\n\npivot_normalized.plot(\n\n    kind='bar',\n\n    stacked=True,\n\n    colormap='coolwarm',\n\n    edgecolor='black',\n\n    ax=ax\n\n)\n\n\n\n# Customizations\n\nplt.title('Distribution of SII Levels by Age Group and Internet Usage Hours', fontsize=20, fontweight='bold')\n\nplt.xlabel('Age Group and Internet Usage Hours', fontsize=16)\n\nplt.ylabel('Proportion of SII Levels', fontsize=16)\n\nplt.xticks(fontsize=12, rotation=45, ha='right')\n\nplt.yticks(fontsize=12)\n\n\n\n# Rename legend entries for clarity\n\nhandles, labels = ax.get_legend_handles_labels()\n\n\n\nlabels = [sii_labels_mapping.get(label, label) for label in labels]\n\n\n\nplt.legend(\n\n    handles=handles,\n\n    labels=labels,\n\n    title='Severity Impairment Index (SII)',\n\n    title_fontsize=12,\n\n    fontsize=10,\n\n    loc='upper left',\n\n    bbox_to_anchor=(1, 1)\n\n)\n\n# save to pdf\n\nplt.savefig('distribution_of_sii_levels_by_age_group_and_internet_usage_hours.pdf', dpi = 600, bbox_inches='tight')\n\n# Layout adjustments\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"XijwpWyOZrhc","executionInfo":{"status":"ok","timestamp":1733665387760,"user_tz":-120,"elapsed":3839,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"49b11436-ccf3-49b0-ae5d-54e4785c50eb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Age Distribution by Severity Impairment Index\n","metadata":{"id":"l26zxYSEgwSk"}},{"cell_type":"code","source":"# Violin plot for age distribution by severity impairment index\n\nplt.figure(figsize=(15, 10))\n\nsns.violinplot(\n\n    data=df_train_csv,\n\n    x='Severity Impairment Index (SII)',\n\n    y='Age of participant',\n\n    palette='coolwarm',\n\n    inner='quartile',\n\n    scale='width'\n\n)\n\nsns.stripplot(\n\n    data=df_train_csv,\n\n    x='Severity Impairment Index (SII)',\n\n    y='Age of participant',\n\n    color='black',\n\n    size=3,\n\n    alpha=0.5,\n\n    jitter=True\n\n)\n\n# update x values to mapped sii\n\nplt.xticks(ticks=[0, 1, 2, 3], labels=['None', 'Mild', 'Moderate', 'Severe'])\n\nplt.title('Age Distribution by Severity Impairment Index', fontsize=18, fontweight='bold', pad=15)\n\nplt.xlabel('Severity Level', fontsize=14)\n\nplt.ylabel('Age', fontsize=14)\n\nplt.xticks(fontsize=12)\n\nplt.yticks(fontsize=12)\n\nsns.despine()\n\n# save\n\nplt.savefig('age_distribution_by_sii.pdf', dpi = 600, bbox_inches='tight')\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"bkpYa57hWC_S","executionInfo":{"status":"ok","timestamp":1733666187594,"user_tz":-120,"elapsed":2543,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"2cf0e4fa-69bd-4922-f2c7-13e22bf4ad4e","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\n\nsummary_stats = df_train_csv.groupby('Severity Impairment Index (SII)')['Age of participant'].agg(['mean']).reset_index()\n\n\n\nsns.barplot(\n\n    data=summary_stats.melt(id_vars='Severity Impairment Index (SII)', var_name='Statistic', value_name='Age'),\n\n    x='Severity Impairment Index (SII)',\n\n    y='Age',\n\n    palette='coolwarm',\n\n\n\n)\n\nplt.xticks(ticks=[0, 1, 2, 3], labels=['None', 'Mild', 'Moderate', 'Severe'])\n\n\n\nplt.title('Averega Age by Severity Level', fontsize=20, fontweight='bold', pad=15)\n\nplt.xlabel('Severity Level', fontsize=16)\n\nplt.ylabel('Age', fontsize=16)\n\nplt.xticks(fontsize=14)\n\nplt.yticks(fontsize=14)\n\nsns.despine()\n\nplt.tight_layout()\n\nplt.show()\n\n\n","metadata":{"id":"yCrKphB_X6xf","executionInfo":{"status":"ok","timestamp":1733666206798,"user_tz":-120,"elapsed":1907,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"a11a1939-b47f-497c-943b-1f9412325ff0","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## SII Distribution by Age and Internet Usage","metadata":{"id":"AUXhMfXuj3ck"}},{"cell_type":"code","source":"# Group data by age and SII level, summing up the counts of internet usage\n\nsii_age_distribution = df_train_csv.groupby(['Age of participant', 'Severity Impairment Index (SII)'])['Hours of using computer/internet'].count().reset_index(name='Count')\n\n\n\n# Pivot the data for stacking\n\nsii_age_pivot = sii_age_distribution.pivot_table(\n\n    index='Age of participant',\n\n    columns='Severity Impairment Index (SII)',\n\n    values='Count',\n\n    fill_value=0\n\n)\n\n# Normalize to get proportions\n\nsii_age_pivot_normalized = sii_age_pivot.div(sii_age_pivot.sum(axis=1), axis=0)\n\nsns.set_style(\"whitegrid\")\n\n# poster\n\nsns.set_context(\"poster\")\n\n# Plot the stacked area chart\n\nsii_age_pivot_normalized.plot(\n\n    kind='area',\n\n    stacked=True,\n\n    colormap='coolwarm',\n\n    alpha=0.8,\n\n    figsize=(16, 10)\n\n)\n\n\n\n# Add title and labels\n\nplt.title('SII Distribution by Age and Internet Usage', fontsize=20, fontweight='bold', pad=20)\n\nplt.xlabel('Age', fontsize=16)\n\nplt.ylabel('Proportion of SII Levels', fontsize=16)\n\nplt.xticks(fontsize=12)\n\nplt.yticks(fontsize=12)\n\nlabels = [sii_labels_mapping.get(label, label) for label in labels]\n\n\n\nplt.legend(\n\n    handles=handles,\n\n    labels=labels,\n\n    title='Severity Impairment Index (SII)',\n\n    title_fontsize=12,\n\n    fontsize=10,\n\n    loc='upper left',\n\n    bbox_to_anchor=(1, 1)\n\n)\n\nplt.grid(color='gray', linestyle='--', linewidth=0.5, alpha=0.7)\n\nsns.despine()\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"wEAJn1U_dB1E","executionInfo":{"status":"ok","timestamp":1733666441003,"user_tz":-120,"elapsed":2182,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"35ea4ae1-1de8-4c1f-b1ff-99bd8f75694e","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Proportional Distribution of SII Levels Across Age Groups","metadata":{"id":"QbXI25tnk6vr"}},{"cell_type":"code","source":"\n\n# Aggregate the data to calculate proportions\n\nage_sii_data = df_train_csv.groupby(['Age of participant', 'Severity Impairment Index (SII)']).size().reset_index(name='Count')\n\n\n\n# Pivot the data\n\nage_sii_pivot = age_sii_data.pivot(index='Age of participant', columns='Severity Impairment Index (SII)', values='Count').fillna(0)\n\n\n\n# Normalize to calculate proportions\n\nage_sii_pivot_normalized = age_sii_pivot.div(age_sii_pivot.sum(axis=1), axis=0)\n\n\n\n# Rename columns for better readability\n\nage_sii_pivot_normalized.columns = ['SII 0', 'SII 1', 'SII 2', 'SII 3']\n\n\n\n# Sort by SII 3 (Severe) proportions\n\nage_sii_pivot_normalized = age_sii_pivot_normalized.sort_values(by='SII 3', ascending=False)\n\n\n\nage_sii_pivot_normalized.plot(\n\n    kind='bar',\n\n    stacked=True,\n\n    colormap='coolwarm',\n\n    edgecolor='black',\n\n    alpha=0.9,figsize=(15, 10\n\n                    )\n\n)\n\n\n\n# Add title and labels\n\nplt.title('Proportional Distribution of SII Levels Across Age Groups', fontsize=20, fontweight='bold', pad=15)\n\nplt.xlabel('Age', fontsize=16, fontweight='bold')\n\nplt.ylabel('Proportion of SII Levels', fontsize=16, fontweight='bold')\n\nplt.xticks(fontsize=12, rotation=45, ha='right')\n\nplt.yticks(fontsize=12)\n\n\n\n# Customize the legend\n\nplt.legend(\n\n    title='SII Level',\n\n    labels=sii_labels_mapping.values(),\n\n    title_fontsize=14,\n\n    fontsize=12,\n\n    loc='upper left',\n\n    bbox_to_anchor=(1, 1)\n\n)\n\n\n\n# Add grid for better readability\n\nplt.grid(axis='y', linestyle='--', linewidth=0.5, alpha=0.7)\n\n\n\n# Adjust layout and show\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"VRClPACddVRA","executionInfo":{"status":"ok","timestamp":1733666825966,"user_tz":-120,"elapsed":1527,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"d439eb1c-6a86-4e3f-cdcb-67402106fc67","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Proportional Distribution of SII Levels by Age Group","metadata":{"id":"3Z5Aq4nWmCzi"}},{"cell_type":"code","source":"sii_age_group = df_train_csv.groupby(['Age Group', 'Severity Impairment Index (SII)']).size().reset_index(name='Count')\n\nsii_age_pivot = sii_age_group.pivot_table(\n\n    index='Age Group',\n\n    columns='Severity Impairment Index (SII)',\n\n    values='Count',\n\n    fill_value=0\n\n)\n\n\n\n# Normalize the data to show proportions\n\nsii_age_pivot_normalized = sii_age_pivot.div(sii_age_pivot.sum(axis=1), axis=0)\n\n\n\n# Rename columns for clarity\n\nsii_age_pivot_normalized.columns = ['SII 0', 'SII 1', 'SII 2', 'SII 3']\n\n\n\n# Plot the grouped bar chart\n\nsii_age_pivot_normalized.plot(\n\n    kind='bar',\n\n    stacked=True,\n\n    figsize=(12, 8),\n\n    colormap='coolwarm',\n\n    edgecolor='black'\n\n)\n\n\n\n# Add labels and title\n\nplt.title('Proportional Distribution of SII Levels by Age Group', fontsize=20, fontweight='bold')\n\nplt.xlabel('Age Group', fontsize=16)\n\nplt.ylabel('Proportion of SII Levels', fontsize=16)\n\nplt.xticks(fontsize=14, rotation=0)\n\nplt.yticks(fontsize=12)\n\nplt.legend(\n\n    title='SII Level',\n\n    fontsize=12,\n\n    title_fontsize=14,\n\n    loc='upper right'\n\n)\n\nplt.tight_layout()\n\nplt.show()\n","metadata":{"id":"Rv38ZYKHdvf3","executionInfo":{"status":"ok","timestamp":1733666779499,"user_tz":-120,"elapsed":2110,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"eaebf730-7601-4a21-e45d-9893519116b4","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## USE PCAIT for Data analysis","metadata":{"id":"PuvrJAQHu3CD"}},{"cell_type":"markdown","source":"## Feature Correlations with SII and Internet Hours","metadata":{"id":"woKri6r9nc1y"}},{"cell_type":"code","source":"# Calculate correlations with SII and Internet Hours\n\ncorrelations = df_train_csv[question_columns + ['Hours of using computer/internet', 'Severity Impairment Index (SII)']].corr()\n\n# map the PCIAT questions to meaningful names\n\nmapped_PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\n# Extract correlations for SII and Internet Hours\n\nsii_corr = correlations['Severity Impairment Index (SII)'].drop(['Hours of using computer/internet', 'Severity Impairment Index (SII)'])\n\ninternet_corr = correlations['Hours of using computer/internet'].drop(['Hours of using computer/internet', 'Severity Impairment Index (SII)'])\n\n\n\n# Combine results into a DataFrame\n\ncorrelation_summary = pd.DataFrame({\n\n    'SII Correlation': sii_corr,\n\n    'Internet Hours Correlation': internet_corr\n\n}).sort_values(by='SII Correlation', ascending=False)\n\nsorted_questions = correlation_summary.index  # Sort rows by SII Correlation\n\nsorted_heatmap = correlations.loc[sorted_questions, ['Severity Impairment Index (SII)', 'Hours of using computer/internet']]\n\n\n\n# Plot heatmap for correlation matrix\n\nplt.figure(figsize=(20, 10))\n\nsns.heatmap(sorted_heatmap,\n\n            annot=True, fmt=\".2f\", cmap='coolwarm', linewidths=1, linecolor='black')\n\nplt.title('Feature Correlations with SII and Internet Hours', fontsize=18, fontweight='bold')\n\n# save\n\nplt.savefig('feature_correlations_with_sii_and_internet_hours.pdf', dpi = 600, bbox_inches='tight')\n\nplt.tight_layout()\n\nplt.show()\n\n\n","metadata":{"id":"ftAjy18dv-FB","executionInfo":{"status":"ok","timestamp":1733667195836,"user_tz":-120,"elapsed":6079,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"ffc92291-6e54-4993-ee25-cfbfe0def3dc","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Do younger children, who are prohibited from accessing the internet, still exhibit addiction-like behaviors through high scores on relevant questions?","metadata":{"id":"4nddXZ5kzus8"}},{"cell_type":"markdown","source":"## Severity Impairment Index and Question Scores Across Age Groups","metadata":{"id":"VxgvoZSfpQkG"}},{"cell_type":"code","source":"\n\n# Map concise labels for relevant questions\n\nrelevent_questions = ['Disobeys Time Limits','Preoccupied With Being Online','Caught Sneaking Online','Angry About Time Limits','Depressed When Offline']\n\n\n\n# Map descriptive labels for Severity Impairment Index (SII)\n\nsii_labels = {0: 'None', 1: 'Mild', 2: 'Moderate', 3: 'Severe'}\n\n\n\n# Custom color palette\n\nsii_palette = {\n\n    'None': 'blue',\n\n    'Mild': '#6baed6',      # Light blue\n\n    'Moderate': '#fd8d3c',  # Orange\n\n    'Severe': 'red'\n\n}\n\n\n\n# Define order for SII\n\nsii_order = ['None', 'Mild', 'Moderate', 'Severe']\n\n\n\n# Select only the relevant columns and rename\n\ndf_selected = df_train_csv[['Age Group',  'Severity Impairment Index (SII)', 'Hours of using computer/internet'] + relevent_questions]\n\n\n\n# Replace SII values with descriptive labels\n\ndf_selected[ 'Severity Impairment Index (SII)'] = df_selected[ 'Severity Impairment Index (SII)'].replace(sii_labels)\n\n\n\n# Sort data to ensure proper order\n\ndf_selected[ 'Severity Impairment Index (SII)'] = pd.Categorical(df_selected[ 'Severity Impairment Index (SII)'], categories=sii_order, ordered=True)\n\n\n\n# Aggregate data: Compute average scores for each question and average internet hours\n\naggregation_functions = {q: 'mean' for q in relevent_questions}\n\naggregation_functions['Hours of using computer/internet'] = 'mean'\n\n\n\naggregated_data = df_selected.groupby(['Age Group',  'Severity Impairment Index (SII)']).agg(aggregation_functions).reset_index()\n\n\n\n# Melt the aggregated data for easier plotting\n\ndf_aggregated_melted = aggregated_data.melt(\n\n    id_vars=['Age Group',  'Severity Impairment Index (SII)', 'Hours of using computer/internet'],\n\n    value_vars=relevent_questions,\n\n    var_name='Question',\n\n    value_name='Average Score'\n\n)\n\n\n\n# Create the FacetGrid plot\n\ng = sns.catplot(\n\n    data=df_aggregated_melted,\n\n    x='Age Group',\n\n    y='Average Score',\n\n    hue= 'Severity Impairment Index (SII)',\n\n    hue_order=sii_order,  # Sort SII levels\n\n    col='Question',\n\n    kind='bar',\n\n    palette=sii_palette,  # Apply custom palette\n\n    height=6,\n\n    aspect=1.2,\n\n    legend_out=True\n\n)\n\n\n\n# Add text annotations for average internet hours above each bar\n\nfor ax, question in zip(g.axes.flat, df_aggregated_melted['Question'].unique()):\n\n    subset = df_aggregated_melted[df_aggregated_melted['Question'] == question]\n\n\n\n    for p, (_, row) in zip(ax.patches, subset.iterrows()):\n\n        x_pos = p.get_x() + p.get_width() / 2\n\n        y_pos = p.get_height()\n\n        internet_hours = row['Hours of using computer/internet']\n\n\n\n        ax.text(\n\n            x=x_pos,\n\n            y=y_pos + 0.05,\n\n            s=f\"{internet_hours:.1f} hrs\",\n\n            ha='center',\n\n            va='bottom',\n\n            fontsize=9,\n\n            color='black',\n\n            fontweight='bold'\n\n        )\n\n\n\n# Ensure Y-axis ticks are clear and granular\n\nfor ax in g.axes.flat:\n\n    ax.set_ylim(0, 5)  # Adjust Y-axis range as needed\n\n    ax.set_yticks(range(0, 6))  # Fixed tick intervals: 0, 1, 2, ..., 5\n\n\n\n# Customize the plot\n\ng.fig.suptitle('SII and Question Scores Across Age Groups with Time Spent Online', fontsize=20, fontweight='bold', y=1.05)\n\ng.set_axis_labels('Age Group', 'Average Question Score')\n\ng.set_titles(col_template='{col_name}', size=14, weight='bold')\n\n# Increase figure size\n\ng.fig.set_size_inches(30, 8)  # Width, Height\n\n\n\n# Adjust spacing between subplots\n\nhandles_labels = g._legend_data.items()\n\nhandles_labels = {(k): v for k, v in handles_labels}\n\nlabels = handles_labels.keys()\n\nhandles = handles_labels.values()\n\n\n\n# Adjust and format the legend\n\ng.legend.remove()  # Remove the default legend\n\nplt.legend(\n\n    handles=handles,\n\n    labels=labels,\n\n    title='Severity Impairment Index (SII)',\n\n    loc='upper left',\n\n    bbox_to_anchor=(1.02, 1),\n\n    fontsize=12,\n\n    title_fontsize=14\n\n)\n\n# Final adjustments\n\n# highlight that we have hours spent as well.\n\n# Add a descriptive note to clarify text annotations\n\n# Remove repeated X-axis labels and keep only the bottom row\n\n\n\n# Add text below the legend\n\n# Add a descriptive note to clarify text annotations\n\ng.fig.text(\n\n    x=0.95, y=0.6,  # Position at the bottom center\n\n    s=\"Text above bars indicates\\n average hours spent online per day\",\n\n    ha='center',\n\n    va='center',\n\n    fontsize=9,\n\n    fontweight='bold',\n\n    color='black'\n\n)\n\n\n\n\n\nplt.tight_layout()\n\nplt.savefig('age_group_question_scores.pdf', dpi=600, bbox_inches='tight')\n\nplt.show()\n","metadata":{"collapsed":true,"id":"5o2Uq_o8nrfC","executionInfo":{"status":"ok","timestamp":1733667566942,"user_tz":-120,"elapsed":7493,"user":{"displayName":"Ahmed Wael","userId":"00145246960982981221"}},"outputId":"05e3d0cf-5eaf-42e6-c336-bf19b419fb85","jupyter":{"outputs_hidden":true},"trusted":true},"outputs":[],"execution_count":null}]}