{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom sklearn.base import BaseEstimator, TransformerMixin\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# display settings\npd.set_option('display.max_columns', 500)\npd.set_option('display.max_rows', 100)\npalette='viridis'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":false,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.314216Z","iopub.execute_input":"2024-12-09T22:32:23.315243Z","iopub.status.idle":"2024-12-09T22:32:23.320487Z","shell.execute_reply.started":"2024-12-09T22:32:23.315199Z","shell.execute_reply":"2024-12-09T22:32:23.319392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper Functions\n### filter_by_instrument, general_info, analyze_categorical,plot_categorical_distributions,analyze_numerical, plot_numerical_distributions\nsource: https://www.kaggle.com/code/ahmedwaelz/complete-eda-and-visualization-for-csv-files#Helper-Functions","metadata":{}},{"cell_type":"code","source":"def filter_by_instrument(df_train, df_dict, instrument_filter):\n\n    \"\"\"\n\n    Filter the training dataset by a specific instrument using the dictionary file.\n\n\n\n    Parameters:\n\n        df_train (pd.DataFrame): The training dataset.\n\n        df_dict (pd.DataFrame): The dictionary file.\n\n        instrument_filter (str): Instrument name to filter columns.\n\n\n\n    Returns:\n\n        pd.DataFrame: Filtered training data for the specified instrument.\n\n        pd.DataFrame: Filtered dictionary for the specified instrument.\n\n    \"\"\"\n\n    df_dict_instrument = df_dict[df_dict['Instrument'] == instrument_filter]\n\n    columns = df_dict_instrument['Field'].tolist()\n\n    df_filtered = df_train[columns]\n\n    return df_filtered, df_dict_instrument\n\n\n\n\n\ndef general_info(df, name=\"Dataset\"):\n\n    \"\"\"\n\n    Display general information about the dataset.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset to analyze.\n\n        name (str): The name of the dataset for display purposes.\n\n    \"\"\"\n\n    print(f\"Summary of {name}:\")\n\n    print(df.info())\n\n    print(\"\\nSummary Statistics:\")\n\n    print(df.describe())\n\n    print(\"\\nMissing Values Percentage:\")\n\n    print(df.isnull().sum() / len(df))\n\n    total_rows = df.shape[0]\n\n\n\n    completely_missing_rows = df[df.isnull().all(axis=1)].shape[0]\n\n    # Total missing values\n\n    total_missing_values = df.isnull().sum().sum()\n\n    percentage_completely_missing_rows = (completely_missing_rows / total_rows) * 100\n\n    print(\"\\n Totally Missing Rows Percentage:\")\n\n    print(f\"{percentage_completely_missing_rows:.2f}\")\n\n\n\ndef analyze_categorical(df, categorical_columns):\n\n    \"\"\"\n\n    Analyze categorical columns by printing value counts.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        categorical_columns (list): List of categorical column names.\n\n    \"\"\"\n\n    for col in categorical_columns:\n\n        print(f\"\\nValue Counts for {col}:\")\n\n        print(df[col].value_counts())\n\n\n\ndef analyze_numerical(df, numerical_columns):\n\n    \"\"\"\n\n    Analyze numerical columns by generating descriptive statistics.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        numerical_columns (list): List of numerical column names.\n\n    \"\"\"\n\n    print(\"\\nDescriptive Statistics for Numerical Columns:\")\n\n    print(df[numerical_columns].describe())\n\n\n\ndef plot_numerical_distributions(df, numerical_columns):\n\n    \"\"\"\n\n    Plot distributions for numerical columns.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        numerical_columns (list): List of numerical column names.\n\n    \"\"\"\n\n    for col in numerical_columns:\n\n        plt.figure(figsize=(8, 4))\n\n        sns.histplot(df[col], kde=True, bins=30)\n\n        plt.title(f'Distribution of {col}')\n\n        plt.show()\n\ndef plot_categorical_distributions(df, categorical_columns):\n\n    \"\"\"\n\n    Plot bar charts for categorical columns.\n\n\n\n    Parameters:\n\n        df (pd.DataFrame): The dataset.\n\n        categorical_columns (list): List of categorical column names.\n\n    \"\"\"\n\n    for col in categorical_columns:\n\n        plt.figure(figsize=(8, 4))\n\n        df[col].value_counts().plot(kind='bar', color='skyblue')\n\n        plt.title(f'Distribution of {col}')\n\n        plt.ylabel('Count')\n\n        plt.xlabel('Categories')\n\n        plt.xticks(rotation=45)\n\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.322727Z","iopub.execute_input":"2024-12-09T22:32:23.323055Z","iopub.status.idle":"2024-12-09T22:32:23.335917Z","shell.execute_reply.started":"2024-12-09T22:32:23.323015Z","shell.execute_reply":"2024-12-09T22:32:23.334882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load CSV Files","metadata":{}},{"cell_type":"code","source":"df_train_csv = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n\ndf_test_csv = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n\ndf_dict_csv = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")\n\nunique_instruments = np.unique(df_dict_csv['Instrument'])\n\nprint(f\"Unique Instruments are\\n  {unique_instruments}\")\n\nprint(f\"The number of Unique Instruments is  {len(unique_instruments)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.337409Z","iopub.execute_input":"2024-12-09T22:32:23.33787Z","iopub.status.idle":"2024-12-09T22:32:23.65814Z","shell.execute_reply.started":"2024-12-09T22:32:23.337824Z","shell.execute_reply":"2024-12-09T22:32:23.657174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\"Instruments' are tests/assesments given to participants in the study.\n\n### Training Features","metadata":{}},{"cell_type":"code","source":"training_features = df_train_csv.columns\ntraining_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.659857Z","iopub.execute_input":"2024-12-09T22:32:23.66014Z","iopub.status.idle":"2024-12-09T22:32:23.667357Z","shell.execute_reply.started":"2024-12-09T22:32:23.660112Z","shell.execute_reply":"2024-12-09T22:32:23.66634Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Metric prefixes and corresponding Assement from CSV files\n* BIA-  | 'Bio-electric Impedance Analysis' |  Measure of key body composition elements, including BMI, fat, muscle, and water content.\n* CGAS_ | 'Children's Global Assessment Scale' | Numeric scale used by mental health clinicians to rate the general functioning of youths under the age of 18.\n* FGC-  | 'FitnessGram Child' | Health related physical fitness assessment measuring five different parameters including aerobic capacity, muscular strength, muscular endurance, flexibility, and body composition.\n* FGC-  | 'FitnessGram Vitals and Treadmill' | Measurements of cardiovascular fitness assessed using the NHANES treadmill protocol.\n* id    | 'Identifier'\n* **PCIAT | 'Parent-Child Internet Addiction Test' | 20-item scale that measures characteristics and behaviors associated with compulsive use of the Internet including compulsivity, escapism, and dependency.**\n* PAC_A | 'Physical Activity Questionnaire (Adolescents)' | Information about children's participation in vigorous activities over the last 7 days.\n* PAC_C | 'Physical Activity Questionnaire (Children)' | Information about children's participation in vigorous activities over the last 7 days.\n* SDS-  | 'Sleep Disturbance Scale' | Scale to categorize sleep disorders in children.\n* Basic_Demos | 'Demographics' | Information about age and sex of participants.\n* Physical-   | 'Physical Measures' | Collection of blood pressure, heart rate, height, weight and waist, and hip measurements.\n* PreInt_ | Internet Use - Number of hours of using computer/internet per day.\n### Parquet File  \n* Actigraphy - Objective measure of ecological physical activity through a research-grade biotracker.\n\n### Note: The target variable 'sii' is derived from the PCIAT assement. The PCIAT assement is NOT available in the test data, so we cannot rely on those fields in our prediction.","metadata":{}},{"cell_type":"markdown","source":"### Test Features","metadata":{}},{"cell_type":"code","source":"test_features = df_test_csv.columns\ntest_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.669653Z","iopub.execute_input":"2024-12-09T22:32:23.669969Z","iopub.status.idle":"2024-12-09T22:32:23.686869Z","shell.execute_reply.started":"2024-12-09T22:32:23.66994Z","shell.execute_reply":"2024-12-09T22:32:23.685803Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are more features in the training set than in the test set. Let's check what's the difference(minus the target 'sii') between the training and testing sets.","metadata":{}},{"cell_type":"code","source":"#columns in training that are not in test set\ndiff_columns = []\nfor feature in training_features:\n    if feature not in test_features:\n        diff_columns.append(feature)        \n# Drop target column\ndiff_columns.remove('sii')\ndiff_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.688043Z","iopub.execute_input":"2024-12-09T22:32:23.688373Z","iopub.status.idle":"2024-12-09T22:32:23.70117Z","shell.execute_reply.started":"2024-12-09T22:32:23.688344Z","shell.execute_reply":"2024-12-09T22:32:23.700214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"'PCIAT' columns are not present in testing set. \n'PCIAT' - Parent Child Internet Addiction Test. The PCIAT is a 20-item scale that measures characteristics and behaviors associated with compulsive use of the Internet that include compulsivity, escapism, and dependency. Questions also assess problems related to addictive use in personal, occupational, and social functioning.\nNote: The PCIAT test was created in 1998 and may be outdated compared to todays standards of internet addiction. (Source: Young, K. S. (1998) Internet addiction: The emergence of a new clinical disorder. CyberPsychology and Behavior, 1(3), 237-244.)\n\nPCIAT assement questions: https://www.healthyplace.com/psychological-tests/parent-child-internet-addiction-test","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Follow 1 ID across train.csv and parquet files. ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv') # Load train file\nfirst_id = train['id'][0] # grab ID of first instance in train file\n\nfirst_id_parquet_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={}'.format(first_id) # parquet path for the first ID\n\npd.read_parquet(first_id_parquet)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.702284Z","iopub.execute_input":"2024-12-09T22:32:23.702935Z","iopub.status.idle":"2024-12-09T22:32:23.834861Z","shell.execute_reply.started":"2024-12-09T22:32:23.702904Z","shell.execute_reply":"2024-12-09T22:32:23.833629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First few ids does not have a parquet file? Lets try going the other way - grab a parquet file ID and try to find it in the CSV","metadata":{}},{"cell_type":"code","source":"parquet_id = '00115b9f' # first parquet ID\nfirst_id_parquet_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={}'.format(parquet_id)\ntrain[train['id'] == parquet_id] # find parquet ID in the csv file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.835825Z","iopub.status.idle":"2024-12-09T22:32:23.836184Z","shell.execute_reply.started":"2024-12-09T22:32:23.836014Z","shell.execute_reply":"2024-12-09T22:32:23.836033Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Check out Parquet file data for same child","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(first_id_parquet_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T22:32:23.837218Z","iopub.status.idle":"2024-12-09T22:32:23.837531Z","shell.execute_reply.started":"2024-12-09T22:32:23.837377Z","shell.execute_reply":"2024-12-09T22:32:23.837394Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Each parquet files represents the 'actigraphy' timeseries data for one child. This data was collected through a wearable electronic device (Accelerometer) that monitored the motion of a child. \n\n\"Accelerometers are wearable devices that measure accelerations of the body segment to which the monitor is attached. The signal is usually filtered and pre-processed by the monitor to obtain activity counts, i.e., accelerations due to body movement.\" \n\n* X, Y, Z represent body segment acceleration in the given dimension (ie X is acceleration along the horizontal dimension, with respect to the body). \n* ENMO(euclidean normal minus one) - squared sum of X, Y, Z minus gravity(9.81 m/s^2)","metadata":{}}]}