{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#                        PROBLEMATIC INTERNET USE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-08T15:54:39.249759Z","iopub.execute_input":"2024-10-08T15:54:39.250411Z","iopub.status.idle":"2024-10-08T15:54:39.261715Z","shell.execute_reply.started":"2024-10-08T15:54:39.25036Z","shell.execute_reply":"2024-10-08T15:54:39.259302Z"}}},{"cell_type":"markdown","source":"\n**The analysis belongs to the data provided by child mind institute and healthy brain network from United States.The study shows the internet usage spent by children for each specific age category and usage variation during seasonal change, and the children physical activities are monitored. ****","metadata":{}},{"cell_type":"markdown","source":"# Data collection:\nData is collected from child mind institute and healthy brain network and a predefined dataset on kaggle. \n\n1. train.csv </br>\n2. test.csv  </br>\n3. data_dictionary.csv  </br>\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:27.973060Z","iopub.execute_input":"2024-12-19T05:48:27.973447Z","iopub.status.idle":"2024-12-19T05:48:29.083638Z","shell.execute_reply.started":"2024-12-19T05:48:27.973413Z","shell.execute_reply":"2024-12-19T05:48:29.082516Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Imputation","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport scipy.stats as sts\nfrom scipy.stats import skew\nfrom scipy.stats import kurtosis\nfrom scipy.stats import trim_mean\nimport seaborn as sns\nimport statistics\nimport warnings","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.085806Z","iopub.execute_input":"2024-12-19T05:48:29.086248Z","iopub.status.idle":"2024-12-19T05:48:29.092717Z","shell.execute_reply.started":"2024-12-19T05:48:29.086204Z","shell.execute_reply":"2024-12-19T05:48:29.091693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#reading input files\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.094209Z","iopub.execute_input":"2024-12-19T05:48:29.094626Z","iopub.status.idle":"2024-12-19T05:48:29.148723Z","shell.execute_reply.started":"2024-12-19T05:48:29.094582Z","shell.execute_reply":"2024-12-19T05:48:29.147660Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.150686Z","iopub.execute_input":"2024-12-19T05:48:29.151038Z","iopub.status.idle":"2024-12-19T05:48:29.157204Z","shell.execute_reply.started":"2024-12-19T05:48:29.150976Z","shell.execute_reply":"2024-12-19T05:48:29.156182Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.158608Z","iopub.execute_input":"2024-12-19T05:48:29.158963Z","iopub.status.idle":"2024-12-19T05:48:29.166554Z","shell.execute_reply.started":"2024-12-19T05:48:29.158933Z","shell.execute_reply":"2024-12-19T05:48:29.165591Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_combined = pd.concat([train.reset_index(drop=True),test.reset_index(drop=True)],axis=1)\n# Drop duplicate columns, keeping the first occurrence\ndf_combined = df_combined.loc[:, ~df_combined.columns.duplicated()]","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.168150Z","iopub.execute_input":"2024-12-19T05:48:29.168471Z","iopub.status.idle":"2024-12-19T05:48:29.183103Z","shell.execute_reply.started":"2024-12-19T05:48:29.168442Z","shell.execute_reply":"2024-12-19T05:48:29.181865Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. id\nDescription: A unique identifier in the dataset.\n2. Basic_Demos-Enroll_Season\nDescription: The season during which the individual enrolled in the study or program (e.g., Fall, Summer).\n3. Basic_Demos-Age\nDescription: The age at the time of enrollment or data collection.\n4. Basic_Demos-Sex\nDescription: The gender of the individual, where 0 = male and 1 = female.\n5. CGAS-Season\nDescription: The season during which the Children's Global Assessment Scale (CGAS) score was assessed during Winter, Summer, fall.\n6. CGAS-CGAS_Score\nDescription: The score which evaluate the overall functioning of a child or adolescent from the Children's Global Assessment Scale.\n7. Physical-Season\nDescription: The season during which physical measurements (e.g., height, weight, BMI) were taken.\n8. Physical-BMI\nDescription: The Body Mass Index (BMI) of the individual, which is a measure of body fat based on height and weight.\n9. Physical-Height\nDescription: The height of the individual, typically measured in centimeters or inches.\n10. Physical-Weight\nDescription: The weight of the individual, usually measured in kilograms or pounds.\n11. PCIAT-PCIAT_18\nDescription: The score or measurement from a specific assessment, potentially related to a cognitive or physical task, conducted in 2018.\n12. PCIAT-PCIAT_19\nDescription: The score or measurement from the same assessment as above, conducted in 2019.\n13. PCIAT-PCIAT_20\nDescription: The score or measurement from the same assessment, conducted in 2020.\n14. PCIAT-PCIAT_Total\nDescription: The total or cumulative score from the PCIAT assessment over the different years.\n15. SDS-Season\nDescription: The season during which the Standardized Developmental Score (SDS) was assessed.\n16. SDS-SDS_Total_Raw\nDescription: The raw score from the Standardized Developmental Score (SDS) assessment, which could represent unadjusted values.\n17. SDS-SDS_Total_T\nDescription: The \"T\" score from the SDS, which might represent a standardized score, where raw data is transformed into a standardized format with a mean and standard deviation.\n18. PreInt_EduHx-Season\nDescription: The season in which educational history data related to the individual (possibly prior to intervention) was collected.\n19. PreInt_EduHx-computerinternet_hoursday\nDescription: The number of hours the individual spends on a computer or using the internet per day before the intervention (possibly an indicator of screen time or technology exposure).\n20. sii\nDescription: This abbreviation is unclear without additional context. It could refer to a specific score, index, or measure related to the individual. It might stand for something like \"Social Interaction Index\" or another domain-specific measure, but further information would be needed to clarify its exact meaning.\n","metadata":{"execution":{"iopub.status.busy":"2024-12-07T20:29:22.173834Z","iopub.execute_input":"2024-12-07T20:29:22.174424Z","iopub.status.idle":"2024-12-07T20:29:22.187077Z","shell.execute_reply.started":"2024-12-07T20:29:22.174356Z","shell.execute_reply":"2024-12-07T20:29:22.185785Z"}}},{"cell_type":"code","source":"df_combined.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.184486Z","iopub.execute_input":"2024-12-19T05:48:29.184885Z","iopub.status.idle":"2024-12-19T05:48:29.245452Z","shell.execute_reply.started":"2024-12-19T05:48:29.184844Z","shell.execute_reply":"2024-12-19T05:48:29.244343Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set pandas to display all columns\npd.set_option('display.max_columns', None)\n# Display the count of missing values in all columns\ndf_combined.isna().sum().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.246557Z","iopub.execute_input":"2024-12-19T05:48:29.246846Z","iopub.status.idle":"2024-12-19T05:48:29.259359Z","shell.execute_reply.started":"2024-12-19T05:48:29.246818Z","shell.execute_reply":"2024-12-19T05:48:29.258050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col_to_find = df_combined.columns[df_combined.applymap(lambda x: isinstance(x, str)).any()]\nif len(col_to_find) > 0:\n print(\"Columns with strings:\", col_to_find)\nelse:\n print(\"There are no strings in any column.\")\n# Select columns with numerical data types (int and float)\n\nnumerical_columns = df_combined.select_dtypes(include=['number']).columns\nif len(numerical_columns) > 0:\n print(\"Columns with numbers:\", numerical_columns)\nelse:\n print(\"There are no strings in any numbers.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.261020Z","iopub.execute_input":"2024-12-19T05:48:29.261383Z","iopub.status.idle":"2024-12-19T05:48:29.367947Z","shell.execute_reply.started":"2024-12-19T05:48:29.261352Z","shell.execute_reply":"2024-12-19T05:48:29.366855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_categorical= df_combined[['id', 'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season','Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n'PAQ_C-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']]\ndf_numerica = df_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii','PAQ_A-PAQ_A_Total','Fitness_Endurance-Time_Sec','Fitness_Endurance-Time_Mins','Fitness_Endurance-Max_Stage']]\n\n# Calculate the percentage of each category in each column\ncategory_percentages = df_categorical.apply(lambda col: col.value_counts(normalize=True) * 100)\ncategory_percentages_nu = df_numerica.apply(lambda col: col.value_counts(normalize=True) * 100)\n\n# Identify columns with more than and less than 50% missing values\nmissing_percentage = df_numerica.isnull().mean() * 100\ncolumns_with_missing_data_g = missing_percentage[missing_percentage > 50]\ncolumns_with_missing_data_l = missing_percentage[missing_percentage < 50]\n\n# Identify columns with more than 50% missing values\nmissing_percentage_nu = df_numerica.isnull().mean() * 100\ncolumns_with_missing_data_nu_g = missing_percentage[missing_percentage > 50]\ncolumns_with_missing_data_nu_l = missing_percentage[missing_percentage < 50]\n\n# Display the results\nprint(\"Category Percentages numerical:\")\nprint(category_percentages_nu)\nprint(category_percentages)\n\nprint(\"\\n Columns with more than 50% missing data:\")\nprint(columns_with_missing_data_g)\nprint(columns_with_missing_data_l)\nprint(columns_with_missing_data_nu_g)\nprint(columns_with_missing_data_nu_l)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.371505Z","iopub.execute_input":"2024-12-19T05:48:29.371914Z","iopub.status.idle":"2024-12-19T05:48:29.432796Z","shell.execute_reply.started":"2024-12-19T05:48:29.371881Z","shell.execute_reply":"2024-12-19T05:48:29.431725Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_combined[['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n       'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone',\n       'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n       'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone',\n       'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total',\n       'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n       'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08',\n       'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n       'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16',\n       'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20',\n       'PCIAT-PCIAT_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',\n       'PreInt_EduHx-computerinternet_hoursday', 'sii']].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.434194Z","iopub.execute_input":"2024-12-19T05:48:29.434595Z","iopub.status.idle":"2024-12-19T05:48:29.448682Z","shell.execute_reply.started":"2024-12-19T05:48:29.434541Z","shell.execute_reply":"2024-12-19T05:48:29.447670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_numerica = df_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii']].median()\ndf_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii']].fillna(df_numerica)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.450152Z","iopub.execute_input":"2024-12-19T05:48:29.450476Z","iopub.status.idle":"2024-12-19T05:48:29.482426Z","shell.execute_reply.started":"2024-12-19T05:48:29.450440Z","shell.execute_reply":"2024-12-19T05:48:29.481401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_numerica = df_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii']].median()\ndf_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii']].fillna(df_numerica)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.483644Z","iopub.execute_input":"2024-12-19T05:48:29.483944Z","iopub.status.idle":"2024-12-19T05:48:29.511510Z","shell.execute_reply.started":"2024-12-19T05:48:29.483915Z","shell.execute_reply":"2024-12-19T05:48:29.510489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_combined.columns)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.512702Z","iopub.execute_input":"2024-12-19T05:48:29.513047Z","iopub.status.idle":"2024-12-19T05:48:29.518713Z","shell.execute_reply.started":"2024-12-19T05:48:29.513006Z","shell.execute_reply":"2024-12-19T05:48:29.517666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_categorical= df_combined[['id', 'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season','Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n'PAQ_C-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']]\ndf_numerica = df_combined[['CGAS-CGAS_Score','Physical-BMI','Physical-Height','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday','sii','PAQ_A-PAQ_A_Total','Fitness_Endurance-Time_Sec','Fitness_Endurance-Time_Mins','Fitness_Endurance-Max_Stage']]\n\n# Calculate the percentage of each category in each column\ncategory_percentages = df_categorical.apply(lambda col: col.value_counts(normalize=True) * 100)\ncategory_percentages_nu = df_numerica.apply(lambda col: col.value_counts(normalize=True) * 100)\n\n# Identify columns with more than and less than 50% missing values\nmissing_percentage = df_numerica.isnull().mean() * 100\ncolumns_with_missing_data_g = missing_percentage[missing_percentage > 50]\ncolumns_with_missing_data_l = missing_percentage[missing_percentage < 50]\n\n# Identify columns with more than 50% missing values\nmissing_percentage_nu = df_numerica.isnull().mean() * 100\ncolumns_with_missing_data_nu_g = missing_percentage[missing_percentage > 50]\ncolumns_with_missing_data_nu_l = missing_percentage[missing_percentage < 50]\n\n# Display the results\nprint(\"Category Percentages numerical:\")\nprint(category_percentages_nu)\nprint(category_percentages)\n\nprint(\"\\n Columns with more than 50% missing data:\")\nprint(columns_with_missing_data_g)\nprint(columns_with_missing_data_l)\nprint(columns_with_missing_data_nu_g)\nprint(columns_with_missing_data_nu_l)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.520120Z","iopub.execute_input":"2024-12-19T05:48:29.520503Z","iopub.status.idle":"2024-12-19T05:48:29.586149Z","shell.execute_reply.started":"2024-12-19T05:48:29.520461Z","shell.execute_reply":"2024-12-19T05:48:29.585111Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_combined.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.587213Z","iopub.execute_input":"2024-12-19T05:48:29.587478Z","iopub.status.idle":"2024-12-19T05:48:29.593627Z","shell.execute_reply.started":"2024-12-19T05:48:29.587452Z","shell.execute_reply":"2024-12-19T05:48:29.592637Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"EXPLANATORY DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"df_combined.describe()","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.594860Z","iopub.execute_input":"2024-12-19T05:48:29.595219Z","iopub.status.idle":"2024-12-19T05:48:29.770688Z","shell.execute_reply.started":"2024-12-19T05:48:29.595176Z","shell.execute_reply":"2024-12-19T05:48:29.769575Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#selected features based on summary statistics\ndf_copy = df_combined[['Basic_Demos-Age','Basic_Demos-Sex','CGAS-CGAS_Score','Physical-Weight','Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_FFM','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:29.771873Z","iopub.execute_input":"2024-12-19T05:48:29.772199Z","iopub.status.idle":"2024-12-19T05:48:29.778941Z","shell.execute_reply.started":"2024-12-19T05:48:29.772170Z","shell.execute_reply":"2024-12-19T05:48:29.777767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\n# Columns to impute\ncolumns_to_impute = ['PAQ_A-PAQ_A_Total', 'Fitness_Endurance-Time_Sec', 'Fitness_Endurance-Time_Mins']\n\n# Initialize KNNImputer\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# Check if all columns to impute exist in the DataFrame\nmissing_columns = [col for col in columns_to_impute if col not in df_copy.columns]\nif missing_columns:\n    print(f\"Missing columns: {missing_columns}\")\nelse:\n    # Apply the imputer to the selected columns\n    df_combined[columns_to_impute] = knn_imputer.fit_transform(df_combined[columns_to_impute])\n    print(\"Imputation completed successfully.\")\n\n# Check if there are any missing values left after imputation\nprint(\"Missing values after imputation:\")\nprint(df_combined.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.780363Z","iopub.execute_input":"2024-12-19T05:48:29.780712Z","iopub.status.idle":"2024-12-19T05:48:29.798250Z","shell.execute_reply.started":"2024-12-19T05:48:29.780679Z","shell.execute_reply":"2024-12-19T05:48:29.797025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import trim_mean\nmean_value = df_copy['Basic_Demos-Age'].mean()\nmedian_value =df_copy['Basic_Demos-Age'].median()\nmode_value = df_copy['Basic_Demos-Age'].mode()\nskewness_value= round(sts.skew(df_copy['Basic_Demos-Age'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['Basic_Demos-Age'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['Basic_Demos-Age'],axis=0, bias=True, fisher=0),1)\n\nlower_bound = df_copy['Basic_Demos-Age'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['Basic_Demos-Age'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['Basic_Demos-Age'] < lower_bound) | (df_copy['Basic_Demos-Age'] > upper_bound)].shape[0]\n\ntrim_propotion= 0.1\ntrimmed_mean= trim_mean(df_copy['Basic_Demos-Age'],trim_propotion)\n\nprint(\"Mean:\" ,mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.799975Z","iopub.execute_input":"2024-12-19T05:48:29.800712Z","iopub.status.idle":"2024-12-19T05:48:29.820167Z","shell.execute_reply.started":"2024-12-19T05:48:29.800664Z","shell.execute_reply":"2024-12-19T05:48:29.819065Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['Basic_Demos-Age'],bins=40,density=True)\nplt.hist(df_copy['Basic_Demos-Age'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['Basic_Demos-Age'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('Basic_Demos-Age')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of Basic_Demos-Age')\n\n# Show plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:29.821391Z","iopub.execute_input":"2024-12-19T05:48:29.821681Z","iopub.status.idle":"2024-12-19T05:48:30.141698Z","shell.execute_reply.started":"2024-12-19T05:48:29.821654Z","shell.execute_reply":"2024-12-19T05:48:30.140588Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['Basic_Demos-Sex'].mean()\nmedian_value =df_copy['Basic_Demos-Sex'].median()\nmode_value = df_copy['Basic_Demos-Sex'].mode()\n\n# Calculate skewness and kurtosis for 'PCIAT-PCIAT_Total' column\nskewness_value = skew(df_copy['Basic_Demos-Sex'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['Basic_Demos-Sex'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['Basic_Demos-Sex'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['Basic_Demos-Sex'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['Basic_Demos-Sex'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['Basic_Demos-Sex'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['Basic_Demos-Sex'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['Basic_Demos-Sex'] < lower_bound) | (df_copy['Basic_Demos-Sex'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['Basic_Demos-Sex'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T05:48:30.143345Z","iopub.execute_input":"2024-12-19T05:48:30.143669Z","iopub.status.idle":"2024-12-19T05:48:30.163400Z","shell.execute_reply.started":"2024-12-19T05:48:30.143638Z","shell.execute_reply":"2024-12-19T05:48:30.162429Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['Basic_Demos-Sex'],bins=40,density=True)\nplt.hist(df_copy['Basic_Demos-Sex'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['Basic_Demos-Sex'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('Basic_Demos-Sex')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of Basic_Demos-Age')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:30.165034Z","iopub.execute_input":"2024-12-19T05:48:30.165461Z","iopub.status.idle":"2024-12-19T05:48:30.514870Z","shell.execute_reply.started":"2024-12-19T05:48:30.165417Z","shell.execute_reply":"2024-12-19T05:48:30.513878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['CGAS-CGAS_Score'].mean()\nmedian_value =df_copy['CGAS-CGAS_Score'].median()\nmode_value = df_copy['CGAS-CGAS_Score'].mode()\n\n# Calculate skewness and kurtosis for 'PCIAT-PCIAT_Total' column\nskewness_value = skew(df_copy['CGAS-CGAS_Score'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['CGAS-CGAS_Score'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['CGAS-CGAS_Score'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['CGAS-CGAS_Score'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['CGAS-CGAS_Score'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['CGAS-CGAS_Score'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['CGAS-CGAS_Score'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['CGAS-CGAS_Score'] < lower_bound) | (df_copy['CGAS-CGAS_Score'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['CGAS-CGAS_Score'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:30.516247Z","iopub.execute_input":"2024-12-19T05:48:30.516640Z","iopub.status.idle":"2024-12-19T05:48:30.540070Z","shell.execute_reply.started":"2024-12-19T05:48:30.516597Z","shell.execute_reply":"2024-12-19T05:48:30.538924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_copy.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:30.541704Z","iopub.execute_input":"2024-12-19T05:48:30.542128Z","iopub.status.idle":"2024-12-19T05:48:30.550458Z","shell.execute_reply.started":"2024-12-19T05:48:30.542084Z","shell.execute_reply":"2024-12-19T05:48:30.549471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\n# Columns to impute\ncolumns_to_impute = ['Physical-Weight','CGAS-CGAS_Score','Physical-Weight','Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_FFM','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw','SDS-SDS_Total_T']\n\n# Initialize KNNImputer\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# Check if all columns to impute exist in the DataFrame\nmissing_columns = [col for col in columns_to_impute if col not in df_copy.columns]\nif missing_columns:\n    print(f\"Missing columns: {missing_columns}\")\nelse:\n    # Apply the imputer to the selected columns\n    df_copy[columns_to_impute] = knn_imputer.fit_transform(df_copy[columns_to_impute])\n    print(\"Imputation completed successfully.\")\n\n# Check if there are any missing values left after imputation\nprint(\"Missing values after imputation:\")\nprint(df_copy.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:30.551622Z","iopub.execute_input":"2024-12-19T05:48:30.551928Z","iopub.status.idle":"2024-12-19T05:48:31.879509Z","shell.execute_reply.started":"2024-12-19T05:48:30.551900Z","shell.execute_reply":"2024-12-19T05:48:31.878362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['Physical-Diastolic_BP'].mean()\nmedian_value =df_copy['Physical-Diastolic_BP'].median()\nmode_value = df_copy['Physical-Diastolic_BP'].mode()\n\n# Calculate skewness and kurtosis for 'PCIAT-PCIAT_Total' column\nskewness_value = skew(df_copy['Physical-Diastolic_BP'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['Physical-Diastolic_BP'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['Physical-Diastolic_BP'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['Physical-Diastolic_BP'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['Physical-Diastolic_BP'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['Physical-Diastolic_BP'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['Physical-Diastolic_BP'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['Physical-Diastolic_BP'] < lower_bound) | (df_copy['Physical-Diastolic_BP'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['Physical-Diastolic_BP'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:31.880953Z","iopub.execute_input":"2024-12-19T05:48:31.881259Z","iopub.status.idle":"2024-12-19T05:48:31.902289Z","shell.execute_reply.started":"2024-12-19T05:48:31.881230Z","shell.execute_reply":"2024-12-19T05:48:31.900948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explanatory Data analysis","metadata":{}},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['Physical-HeartRate'].mean()\nmedian_value =df_copy['Physical-HeartRate'].median()\nmode_value = df_copy['Physical-HeartRate'].mode()\n\n# Calculate skewness and kurtosis for 'PCIAT-PCIAT_Total' column\nskewness_value = skew(df_copy['Physical-HeartRate'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['Physical-HeartRate'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['Physical-HeartRate'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['Physical-HeartRate'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['Physical-HeartRate'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['Physical-HeartRate'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['Physical-HeartRate'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['Physical-HeartRate'] < lower_bound) | (df_copy['Physical-HeartRate'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['Physical-HeartRate'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:31.907349Z","iopub.execute_input":"2024-12-19T05:48:31.907683Z","iopub.status.idle":"2024-12-19T05:48:31.928169Z","shell.execute_reply.started":"2024-12-19T05:48:31.907651Z","shell.execute_reply":"2024-12-19T05:48:31.926947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['Physical-HeartRate'],bins=40,density=True)\nplt.hist(df_copy['Physical-HeartRate'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['Physical-HeartRate'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('Physical-HeartRate')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of Physical-HeartRate')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:31.929403Z","iopub.execute_input":"2024-12-19T05:48:31.929750Z","iopub.status.idle":"2024-12-19T05:48:32.511532Z","shell.execute_reply.started":"2024-12-19T05:48:31.929707Z","shell.execute_reply":"2024-12-19T05:48:32.510475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['Physical-Systolic_BP'].mean()\nmedian_value =df_copy['Physical-Systolic_BP'].median()\nmode_value = df_copy['Physical-Systolic_BP'].mode()\n\n# Calculate skewness and kurtosis for 'Physical-Systolic_BP' column\nskewness_value = skew(df_copy['Physical-Systolic_BP'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['Physical-Systolic_BP'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['Physical-Systolic_BP'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['Physical-Systolic_BP'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['Physical-Systolic_BP'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['Physical-Systolic_BP'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['Physical-Systolic_BP'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['Physical-Systolic_BP'] < lower_bound) | (df_copy['Physical-Systolic_BP'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['Physical-Systolic_BP'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:32.512809Z","iopub.execute_input":"2024-12-19T05:48:32.513145Z","iopub.status.idle":"2024-12-19T05:48:32.535697Z","shell.execute_reply.started":"2024-12-19T05:48:32.513114Z","shell.execute_reply":"2024-12-19T05:48:32.534525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['Physical-Systolic_BP'],bins=40,density=True)\nplt.hist(df_copy['Physical-Systolic_BP'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['Physical-Systolic_BP'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('Physical-Systolic_BP')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of Physical-Systolic_BP')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:32.536964Z","iopub.execute_input":"2024-12-19T05:48:32.537357Z","iopub.status.idle":"2024-12-19T05:48:32.857314Z","shell.execute_reply.started":"2024-12-19T05:48:32.537325Z","shell.execute_reply":"2024-12-19T05:48:32.856071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['BIA-BIA_BMR'].mean()\nmedian_value =df_copy['BIA-BIA_BMR'].median()\nmode_value = df_copy['BIA-BIA_BMR'].mode()\n\n# Calculated skewness and kurtosis for 'BIA-BIA_BMR' column\nskewness_value = skew(df_copy['BIA-BIA_BMR'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['BIA-BIA_BMR'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['BIA-BIA_BMR'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['BIA-BIA_BMR'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['BIA-BIA_BMR'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['BIA-BIA_BMR'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['BIA-BIA_BMR'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['BIA-BIA_BMR'] < lower_bound) | (df_copy['BIA-BIA_BMR'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['BIA-BIA_BMR'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:32.858682Z","iopub.execute_input":"2024-12-19T05:48:32.859123Z","iopub.status.idle":"2024-12-19T05:48:32.883850Z","shell.execute_reply.started":"2024-12-19T05:48:32.859079Z","shell.execute_reply":"2024-12-19T05:48:32.882690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['BIA-BIA_BMR'],bins=40,density=True)\nplt.hist(df_copy['BIA-BIA_BMR'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['BIA-BIA_BMR'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('BIA-BIA_BMR')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of BIA-BIA_BMR')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:32.885180Z","iopub.execute_input":"2024-12-19T05:48:32.885481Z","iopub.status.idle":"2024-12-19T05:48:33.240511Z","shell.execute_reply.started":"2024-12-19T05:48:32.885450Z","shell.execute_reply":"2024-12-19T05:48:33.239322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['BIA-BIA_DEE'].mean()\nmedian_value =df_copy['BIA-BIA_DEE'].median()\nmode_value = df_copy['BIA-BIA_DEE'].mode()\n\n# Calculate skewness and kurtosis for 'BIA-BIA_DEE' column\nskewness_value = skew(df_copy['BIA-BIA_DEE'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['BIA-BIA_DEE'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['BIA-BIA_DEE'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['BIA-BIA_DEE'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['BIA-BIA_DEE'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['BIA-BIA_DEE'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['BIA-BIA_DEE'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['BIA-BIA_DEE'] < lower_bound) | (df_copy['BIA-BIA_DEE'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['BIA-BIA_DEE'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:33.242023Z","iopub.execute_input":"2024-12-19T05:48:33.242432Z","iopub.status.idle":"2024-12-19T05:48:33.269211Z","shell.execute_reply.started":"2024-12-19T05:48:33.242387Z","shell.execute_reply":"2024-12-19T05:48:33.268168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['BIA-BIA_DEE'],bins=40,density=True)\nplt.hist(df_copy['BIA-BIA_DEE'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['BIA-BIA_DEE'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('BIA-BIA_DEE')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of BIA-BIA_DEE')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:33.271079Z","iopub.execute_input":"2024-12-19T05:48:33.271511Z","iopub.status.idle":"2024-12-19T05:48:33.636905Z","shell.execute_reply.started":"2024-12-19T05:48:33.271466Z","shell.execute_reply":"2024-12-19T05:48:33.635813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['BIA-BIA_FFM'].mean()\nmedian_value =df_copy['BIA-BIA_FFM'].median()\nmode_value = df_copy['BIA-BIA_FFM'].mode()\n\n# Calculate skewness and kurtosis for 'BIA-BIA_FFM' column\nskewness_value = skew(df_copy['BIA-BIA_FFM'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['BIA-BIA_FFM'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['BIA-BIA_FFM'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['BIA-BIA_FFM'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['BIA-BIA_FFM'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['BIA-BIA_FFM'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['BIA-BIA_FFM'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['BIA-BIA_FFM'] < lower_bound) | (df_copy['BIA-BIA_FFM'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['BIA-BIA_FFM'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:33.637865Z","iopub.execute_input":"2024-12-19T05:48:33.638158Z","iopub.status.idle":"2024-12-19T05:48:33.660083Z","shell.execute_reply.started":"2024-12-19T05:48:33.638131Z","shell.execute_reply":"2024-12-19T05:48:33.659034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['BIA-BIA_FFM'],bins=40,density=True)\nplt.hist(df_copy['BIA-BIA_FFM'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['BIA-BIA_FFM'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('BIA-BIA_FFM')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of BIA-BIA_FFM')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:33.661565Z","iopub.execute_input":"2024-12-19T05:48:33.662016Z","iopub.status.idle":"2024-12-19T05:48:34.008133Z","shell.execute_reply.started":"2024-12-19T05:48:33.661947Z","shell.execute_reply":"2024-12-19T05:48:34.007057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['PCIAT-PCIAT_Total'].mean()\nmedian_value =df_copy['PCIAT-PCIAT_Total'].median()\nmode_value = df_copy['PCIAT-PCIAT_Total'].mode()\n\n# Calculate skewness and kurtosis for 'PCIAT-PCIAT_Total' column\nskewness_value = skew(df_copy['PCIAT-PCIAT_Total'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['PCIAT-PCIAT_Total'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['PCIAT-PCIAT_Total'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['PCIAT-PCIAT_Total'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['PCIAT-PCIAT_Total'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['PCIAT-PCIAT_Total'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['PCIAT-PCIAT_Total'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['PCIAT-PCIAT_Total'] < lower_bound) | (df_copy['PCIAT-PCIAT_Total'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['PCIAT-PCIAT_Total'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.009612Z","iopub.execute_input":"2024-12-19T05:48:34.010090Z","iopub.status.idle":"2024-12-19T05:48:34.037124Z","shell.execute_reply.started":"2024-12-19T05:48:34.010043Z","shell.execute_reply":"2024-12-19T05:48:34.036073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['PCIAT-PCIAT_Total'],bins=40,density=True)\nplt.hist(df_copy['PCIAT-PCIAT_Total'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['PCIAT-PCIAT_Total'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('PCIAT-PCIAT_Total')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of PCIAT-PCIAT_Total')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.038454Z","iopub.execute_input":"2024-12-19T05:48:34.038791Z","iopub.status.idle":"2024-12-19T05:48:34.381625Z","shell.execute_reply.started":"2024-12-19T05:48:34.038760Z","shell.execute_reply":"2024-12-19T05:48:34.380545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['SDS-SDS_Total_Raw'].mean()\nmedian_value =df_copy['SDS-SDS_Total_Raw'].median()\nmode_value = df_copy['SDS-SDS_Total_Raw'].mode()\n\n# Calculate skewness and kurtosis for 'SDS-SDS_Total_Raw' column\nskewness_value = skew(df_copy['SDS-SDS_Total_Raw'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['SDS-SDS_Total_Raw'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['SDS-SDS_Total_Raw'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['SDS-SDS_Total_Raw'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['SDS-SDS_Total_Raw'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['SDS-SDS_Total_Raw'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['SDS-SDS_Total_Raw'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['SDS-SDS_Total_Raw'] < lower_bound) | (df_copy['SDS-SDS_Total_Raw'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['SDS-SDS_Total_Raw'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.383100Z","iopub.execute_input":"2024-12-19T05:48:34.383521Z","iopub.status.idle":"2024-12-19T05:48:34.410883Z","shell.execute_reply.started":"2024-12-19T05:48:34.383476Z","shell.execute_reply":"2024-12-19T05:48:34.409925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['SDS-SDS_Total_Raw'],bins=40,density=True)\nplt.hist(df_copy['SDS-SDS_Total_Raw'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['SDS-SDS_Total_Raw'])\n\n# Generate x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculate normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Plot normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Add labels, title, and legend\nplt.xlabel('SDS-SDS_Total_Raw')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of SDS-SDS_Total_Raw')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.412445Z","iopub.execute_input":"2024-12-19T05:48:34.412791Z","iopub.status.idle":"2024-12-19T05:48:34.736641Z","shell.execute_reply.started":"2024-12-19T05:48:34.412759Z","shell.execute_reply":"2024-12-19T05:48:34.735553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis  # Ensure you import the functions\nfrom scipy.stats import trim_mean\ntrim_propotion= 0.1\nmean_value = df_copy['SDS-SDS_Total_T'].mean()\nmedian_value =df_copy['SDS-SDS_Total_T'].median()\nmode_value = df_copy['SDS-SDS_Total_T'].mode()\n\n# Calculated skewness and kurtosis for 'SDS-SDS_Total_T' column\nskewness_value = skew(df_copy['SDS-SDS_Total_T'], nan_policy='omit')  # Ignore NaNs if any\nkurtosis_value = kurtosis(df_copy['SDS-SDS_Total_T'], nan_policy='omit')  # Ignore NaNs if any\n\nskewness_value= round(sts.skew(df_copy['SDS-SDS_Total_T'],axis=0, bias=True),1)\nIQR = round(sts.iqr(df_copy['SDS-SDS_Total_T'],axis=0,rng=(25,75)),2)\nkurtosis = round(sts.kurtosis(df_copy['SDS-SDS_Total_T'],axis=0, bias=True, fisher=0),1)\nlower_bound = df_copy['SDS-SDS_Total_T'].quantile(0.25) - 1.5 * IQR\nupper_bound = df_copy['SDS-SDS_Total_T'].quantile(0.75) + 1.5 * IQR\n\n# Identify outliers\noutliers_count = df_copy[(df_copy['SDS-SDS_Total_T'] < lower_bound) | (df_copy['SDS-SDS_Total_T'] > upper_bound)].shape[0]\n\ntrimmed_mean= trim_mean(df_copy['SDS-SDS_Total_T'],trim_propotion)\nprint(\"Mean:\",mean_value)\nprint(\"Median:\",median_value)\nprint(\"Mode:\",mode_value)\nprint(\"Skewness:\",skewness_value)\nprint(\"IQR:\",IQR)\nprint(\"kurtosis:\",kurtosis)\nprint(\"outlier:\", outliers_count)\nprint(\"trimmed_mean:\",trimmed_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.737889Z","iopub.execute_input":"2024-12-19T05:48:34.738216Z","iopub.status.idle":"2024-12-19T05:48:34.760309Z","shell.execute_reply.started":"2024-12-19T05:48:34.738184Z","shell.execute_reply":"2024-12-19T05:48:34.759020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#figure size\nplt.rcParams['figure.figsize'] = (8, 6)\n\n#specifying histogram and bins\nhist,bin_edges = np.histogram(df_copy['SDS-SDS_Total_T'],bins=40,density=True)\nplt.hist(df_copy['SDS-SDS_Total_T'], bins=40, density=True, edgecolor='grey', alpha=0.4)\n\nplt.axvline(mean_value, color='r', linestyle='dashed',linewidth=1, label= f'Mean:{mean_value:.2f}')\nplt.axvline(median_value,color='g',linestyle='dashed',linewidth=1,label= f'Median:{ median_value:.2f}')\nplt.axvline(trimmed_mean,color='y',linestyle='dashed',linewidth=1,label= f'Trimmedmean:{trimmed_mean:.2f}')\n\nmn,std = sts.norm.fit(df_copy['SDS-SDS_Total_T'])\n\n# Generated x values for the normal distribution curve\nx_bin = np.linspace(bin_edges[0], bin_edges[-1], 100)\n\n# Calculated normal distribution curve\ny_curve = sts.norm.pdf(x_bin, mn, std)\n\n# Ploted normal distribution curve\nplt.plot(x_bin, y_curve, 'k', linewidth=2, label='Normal Distribution Fit')\n\n# Added labels, title, and legend\nplt.xlabel('SDS-SDS_Total_T')\nplt.ylabel('Density')\nplt.legend()\nplt.title('Histogram of SDS-SDS_Total_T')\n\n# Show plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:34.761683Z","iopub.execute_input":"2024-12-19T05:48:34.762068Z","iopub.status.idle":"2024-12-19T05:48:35.086561Z","shell.execute_reply.started":"2024-12-19T05:48:34.762035Z","shell.execute_reply":"2024-12-19T05:48:35.085436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"mode < median < mean from the above visualization, it is confirmed that the distribution is non-normal in nature. ","metadata":{}},{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"print(df_combined['sii'].value_counts(normalize=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:35.087801Z","iopub.execute_input":"2024-12-19T05:48:35.088129Z","iopub.status.idle":"2024-12-19T05:48:35.095258Z","shell.execute_reply.started":"2024-12-19T05:48:35.088099Z","shell.execute_reply":"2024-12-19T05:48:35.094202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The dataset is highly imbalanced due to the social interaction scale in this study is higher compare to the non-social interaction with humans.","metadata":{}},{"cell_type":"code","source":"\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:48:35.126104Z","iopub.execute_input":"2024-12-19T05:48:35.126399Z","iopub.status.idle":"2024-12-19T05:48:35.134181Z","shell.execute_reply.started":"2024-12-19T05:48:35.126371Z","shell.execute_reply":"2024-12-19T05:48:35.132873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}