{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Preprocessing","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-28T13:58:55.324188Z","iopub.execute_input":"2024-10-28T13:58:55.324599Z","iopub.status.idle":"2024-10-28T13:58:58.942837Z","shell.execute_reply.started":"2024-10-28T13:58:55.324557Z","shell.execute_reply":"2024-10-28T13:58:58.941476Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.758986Z","iopub.execute_input":"2024-11-03T23:46:06.759458Z","iopub.status.idle":"2024-11-03T23:46:06.765302Z","shell.execute_reply.started":"2024-11-03T23:46:06.759414Z","shell.execute_reply":"2024-11-03T23:46:06.764050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndictionary","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.777611Z","iopub.execute_input":"2024-11-03T23:46:06.778790Z","iopub.status.idle":"2024-11-03T23:46:06.797982Z","shell.execute_reply.started":"2024-11-03T23:46:06.778733Z","shell.execute_reply":"2024-11-03T23:46:06.796840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.799721Z","iopub.execute_input":"2024-11-03T23:46:06.800121Z","iopub.status.idle":"2024-11-03T23:46:06.880042Z","shell.execute_reply.started":"2024-11-03T23:46:06.800082Z","shell.execute_reply":"2024-11-03T23:46:06.878935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop data if there is no sii\nThere is no reference","metadata":{}},{"cell_type":"code","source":"column_to_check = 'sii'\n\ntrain_cleaned = train.dropna(subset=[column_to_check])\n\ntrain_cleaned.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.882383Z","iopub.execute_input":"2024-11-03T23:46:06.882821Z","iopub.status.idle":"2024-11-03T23:46:06.919686Z","shell.execute_reply.started":"2024-11-03T23:46:06.882771Z","shell.execute_reply":"2024-11-03T23:46:06.918653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Physical Measures","metadata":{}},{"cell_type":"markdown","source":"### Drop the data without Physical-BMI\nJust assume that they are dead","metadata":{}},{"cell_type":"code","source":"column_to_check = 'Physical-BMI'\n\ntrain_cleaned = train.dropna(subset=[column_to_check])\n\ntrain_cleaned.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.921001Z","iopub.execute_input":"2024-11-03T23:46:06.921356Z","iopub.status.idle":"2024-11-03T23:46:06.957553Z","shell.execute_reply.started":"2024-11-03T23:46:06.921320Z","shell.execute_reply":"2024-11-03T23:46:06.956446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop data if Physical-Diastolic_BP is >90 and <50\n\nnormal: 53-80 <br>\nif the blood pressure is >90 or <50, you should go to the hospital first","metadata":{}},{"cell_type":"code","source":"column_to_check = 'Physical-Diastolic_BP'\n\ntrain_cleaned = train_cleaned[(train_cleaned[column_to_check] >= 50) & (train_cleaned[column_to_check] <= 90)]\n\ntrain_cleaned[column_to_check].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.960112Z","iopub.execute_input":"2024-11-03T23:46:06.960539Z","iopub.status.idle":"2024-11-03T23:46:06.973524Z","shell.execute_reply.started":"2024-11-03T23:46:06.960501Z","shell.execute_reply":"2024-11-03T23:46:06.972411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop data if heartrate is <55 or >125\n\nschool age(6-12): 70-100 <br>\nadolescents(13-18): 60-100 <br>\nadults(18up): 60-100 <br>\n\nif it is not in this range, you should go to the hospital","metadata":{}},{"cell_type":"code","source":"column_to_check = 'Physical-HeartRate'\n\ntrain_cleaned = train_cleaned[(train_cleaned[column_to_check] >= 55) & (train_cleaned[column_to_check] <= 125)]\n\ntrain_cleaned[column_to_check].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.974849Z","iopub.execute_input":"2024-11-03T23:46:06.975252Z","iopub.status.idle":"2024-11-03T23:46:06.993761Z","shell.execute_reply.started":"2024-11-03T23:46:06.975211Z","shell.execute_reply":"2024-11-03T23:46:06.992394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop data if Physical-Systolic_BP is <90 or >135\n\nnormal: 95-130 <br>\nif the blood pressure is <90 or >135, you should go to the hospital first","metadata":{}},{"cell_type":"code","source":"column_to_check = 'Physical-Systolic_BP'\n\ntrain_cleaned = train_cleaned[(train_cleaned[column_to_check] >= 90) & (train_cleaned[column_to_check] <= 135)]\n\ntrain_cleaned[column_to_check].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:06.995393Z","iopub.execute_input":"2024-11-03T23:46:06.995888Z","iopub.status.idle":"2024-11-03T23:46:07.008766Z","shell.execute_reply.started":"2024-11-03T23:46:06.995849Z","shell.execute_reply":"2024-11-03T23:46:07.007507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"| Age Group              | Systolic       | Diastolic      |\n|------------------------|----------------|----------------|\n| Newborns up to 1 month | 60–90 mm Hg    | 20–60 mm Hg    |\n| Infants                | 87–105 mm Hg   | 53–66 mm Hg    |\n| Toddlers               | 95–105 mm Hg   | 53–66 mm Hg    |\n| Preschoolers           | 95–110 mm Hg   | 56–70 mm Hg    |\n| School-aged children   | 97–112 mm Hg   | 57–71 mm Hg    |\n| Adolescents            | 112–128 mm Hg  | 66–80 mm Hg    |\n\n\n<br>\nsrc: https://www.baptisthealth.com/blog/heart-care/healthy-blood-pressure-by-age-and-gender-chart","metadata":{}},{"cell_type":"markdown","source":"### Drop data if CGAS score is <0 or >100\n\n-> this is by definition, so if CGAS is out of range, it should delete it.","metadata":{}},{"cell_type":"code","source":"column_to_check = 'CGAS-CGAS_Score'\n\ntrain_cleaned = train_cleaned[(train_cleaned[column_to_check] >= 0) & (train_cleaned[column_to_check] <= 100)]\n\ntrain_cleaned[column_to_check].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.010433Z","iopub.execute_input":"2024-11-03T23:46:07.010795Z","iopub.status.idle":"2024-11-03T23:46:07.023488Z","shell.execute_reply.started":"2024-11-03T23:46:07.010758Z","shell.execute_reply":"2024-11-03T23:46:07.022192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# FGC","metadata":{}},{"cell_type":"markdown","source":"### FGC_GSND is sparse\n### FGC_GSD is sparse","metadata":{}},{"cell_type":"markdown","source":"### push up seems OK","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# BIA","metadata":{}},{"cell_type":"markdown","source":"### Drop BIA-BIA_BMC if score is <-1\n\nnormal: -1 or higher<br>\nosteoporosis: -2.5 or lower<br>","metadata":{}},{"cell_type":"code","source":"column_to_check = 'BIA-BIA_BMC'\n\ntrain_cleaned = train_cleaned[train_cleaned[column_to_check]>=-1]\n\ntrain_cleaned[column_to_check].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.025031Z","iopub.execute_input":"2024-11-03T23:46:07.025423Z","iopub.status.idle":"2024-11-03T23:46:07.036819Z","shell.execute_reply.started":"2024-11-03T23:46:07.025373Z","shell.execute_reply":"2024-11-03T23:46:07.035463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### BIA-BIA_BMR is calculated according to a equation\n\nBMR: Basal Metabolic Rate\n\nIt is calculated according to weight, height and age\n\nFormula: <br>\nMen: BMR = 5 + (10 x wt in kg) + (6.25 x ht in cm) - (5 x age in years) <br>\nWomen: BMR = 655 + (9.6 x wt in kg) + (1.8 x ht in cm) - (4.7 x age in years)  <br>","metadata":{}},{"cell_type":"markdown","source":"### BIA-BIA_DEE is calculated according to a equation\n\nDEE: Daily Energy Expenditure\n\nequation: BMR x Activity Factor\n\nSedentary = BMR x 1.2 (little or no exercise, desk job) <br> \nLightly active = BMR x 1.375 (light exercise/ sports 1-3 days/week)<br>\nModerately active = BMR x 1.55 (moderate exercise/ sports 6-7 days/week)<br>\nVery active = BMR x 1.725 (hard exercise every day, or exercising 2 xs/day)<br>\nExtra active = BMR x 1.9 (hard exercise 2 or more times per day, or training for marathon, or triathlon, etc. <br> ","metadata":{}},{"cell_type":"markdown","source":"### BIA-BIA_FMI some are less than 0\nnormal range: Male: 18%-24% Female: 25%-31%","metadata":{}},{"cell_type":"code","source":"column_to_check = 'BIA-BIA_FMI'\n\ntrain_cleaned = train_cleaned[train_cleaned[column_to_check]>=0]\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.039872Z","iopub.execute_input":"2024-11-03T23:46:07.040232Z","iopub.status.idle":"2024-11-03T23:46:07.047864Z","shell.execute_reply.started":"2024-11-03T23:46:07.040193Z","shell.execute_reply":"2024-11-03T23:46:07.046775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### BIA-BIA_Fat some are less than 0\n\nnormal range: Male: 25-30 Female: 30-35","metadata":{}},{"cell_type":"code","source":"column_to_check = 'BIA-BIA_Fat'\n\ntrain_cleaned = train_cleaned[(train_cleaned[column_to_check] >= 15) & (train_cleaned[column_to_check] <= 45)]","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.049379Z","iopub.execute_input":"2024-11-03T23:46:07.049895Z","iopub.status.idle":"2024-11-03T23:46:07.060811Z","shell.execute_reply.started":"2024-11-03T23:46:07.049842Z","shell.execute_reply":"2024-11-03T23:46:07.059646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# PCIAT combination test","metadata":{}},{"cell_type":"markdown","source":"### Here is some experience (temporarily)\n","metadata":{}},{"cell_type":"code","source":"train_cleaned.to_csv('train_cleaned.csv')","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.062842Z","iopub.execute_input":"2024-11-03T23:46:07.063195Z","iopub.status.idle":"2024-11-03T23:46:07.119985Z","shell.execute_reply.started":"2024-11-03T23:46:07.063157Z","shell.execute_reply":"2024-11-03T23:46:07.118815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = train_cleaned","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.122358Z","iopub.execute_input":"2024-11-03T23:46:07.122792Z","iopub.status.idle":"2024-11-03T23:46:07.130639Z","shell.execute_reply.started":"2024-11-03T23:46:07.122740Z","shell.execute_reply":"2024-11-03T23:46:07.129130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'test' is your DataFrame\ndf = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Convert values to integers (if not already)\ndf[columns] = df[columns].astype(int)\n\n# Calculate the correlation matrix\ncorrelation_matrix = df.corr()\n\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Draw the heatmap\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap='Greens', \n            square=True, cbar_kws={\"shrink\": .8}, linewidths=.5)\n\n# Customize the plot\nplt.title('Correlation Heatmap of PCIAT Scores', fontsize=16)\nplt.xticks(rotation=45)\nplt.yticks(rotation=45)\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T00:18:34.491984Z","iopub.execute_input":"2024-11-04T00:18:34.492462Z","iopub.status.idle":"2024-11-04T00:18:36.173607Z","shell.execute_reply.started":"2024-11-04T00:18:34.492419Z","shell.execute_reply":"2024-11-04T00:18:36.172546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.cluster.hierarchy import linkage, dendrogram\nfrom scipy.spatial.distance import squareform\n\n# Assuming 'test' is your DataFrame\ndf = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Convert values to integers (if not already)\ndf[columns] = df[columns].astype(int)\n\n# Calculate the correlation matrix\ncorrelation_matrix = df.corr()\n\n# Convert the correlation matrix to a distance matrix\n# 1 - correlation gives a distance matrix\ndistance_matrix = 1 - correlation_matrix\n\n# Perform hierarchical clustering\nlinkage_matrix = linkage(squareform(distance_matrix), method='ward')\n\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Create a dendrogram to visualize the clusters\ndendrogram(linkage_matrix, labels=correlation_matrix.columns, leaf_rotation=90)\n\nplt.title('Hierarchical Clustering Dendrogram')\nplt.xlabel('Variables')\nplt.ylabel('Distance')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T00:18:38.386970Z","iopub.execute_input":"2024-11-04T00:18:38.387975Z","iopub.status.idle":"2024-11-04T00:18:39.086186Z","shell.execute_reply.started":"2024-11-04T00:18:38.387925Z","shell.execute_reply":"2024-11-04T00:18:39.084951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['PCIAT-PCIAT_Score'] = test['PCIAT-PCIAT_01']+test['PCIAT-PCIAT_02']+test['PCIAT-PCIAT_03']+test['PCIAT-PCIAT_05']\ntest['PCIAT-PCIAT_Score_left'] = test['PCIAT-PCIAT_Total'] - test['PCIAT-PCIAT_Score']\ntest['PCIAT-PCIAT_Score_square'] = round(np.sqrt(np.sqrt((test['PCIAT-PCIAT_01']+1)*(test['PCIAT-PCIAT_02']+1)*(test['PCIAT-PCIAT_03']+1)*(test['PCIAT-PCIAT_05']+1))))-1","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.133403Z","iopub.execute_input":"2024-11-03T23:46:07.133824Z","iopub.status.idle":"2024-11-03T23:46:07.150816Z","shell.execute_reply.started":"2024-11-03T23:46:07.133769Z","shell.execute_reply":"2024-11-03T23:46:07.149449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the range and columns\ncolumns = [f\"PCIAT-PCIAT_{j:02d}\" for j in range(1, 21)]\n\n# Calculate the product across these columns\ntest['sii_product'] = (test[columns]+1).prod(axis=1)\n\n# Apply the n-th root (replace 'n' with the root degree you need)\ntest['sii_square'] = round((test['sii_product']) ** (1 / 20))-1","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.153068Z","iopub.execute_input":"2024-11-03T23:46:07.153481Z","iopub.status.idle":"2024-11-03T23:46:07.167132Z","shell.execute_reply.started":"2024-11-03T23:46:07.153441Z","shell.execute_reply":"2024-11-03T23:46:07.165762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['PCIAT-PCIAT_Score_square']","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.168455Z","iopub.execute_input":"2024-11-03T23:46:07.168831Z","iopub.status.idle":"2024-11-03T23:46:07.184545Z","shell.execute_reply.started":"2024-11-03T23:46:07.168790Z","shell.execute_reply":"2024-11-03T23:46:07.182529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jittered_sii = [s + np.random.uniform(-0.1, 0.1) for s in test['sii']]\n\nplt.figure(figsize=(10, 6))\nplt.scatter(jittered_sii, test['PCIAT-PCIAT_Score'], color='blue', alpha=0.6, edgecolors='w', s=100)\n\n# Set titles and labels\nplt.title('Score Distribution across SII Categories with Jitter')\nplt.xlabel('SII')\nplt.ylabel('Score')\nplt.xticks([0, 1, 2, 3], ['None', 'Mild', 'Moderate', 'Severe'])\nplt.show()\n\nplt.figure(figsize=(10, 6))\nplt.scatter(jittered_sii, test['PCIAT-PCIAT_Score_square'], color='blue', alpha=0.6, edgecolors='w', s=100)\n\n# Set titles and labels\nplt.title('Score Distribution across SII Categories with Jitter')\nplt.xlabel('SII')\nplt.ylabel('Score')\nplt.xticks([0, 1, 2, 3], ['None', 'Mild', 'Moderate', 'Severe'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.188173Z","iopub.execute_input":"2024-11-03T23:46:07.188888Z","iopub.status.idle":"2024-11-03T23:46:07.803475Z","shell.execute_reply.started":"2024-11-03T23:46:07.188843Z","shell.execute_reply":"2024-11-03T23:46:07.802146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#question for all\nimport matplotlib.pyplot as plt\n\n# Define the labels for SII categories and PCIAT-PCIAT_Score_square values\nsii_labels = ['None', 'Mild', 'Moderate', 'Severe']\nscore_labels = [0, 1, 2, 3, 4, 5]\n\n# Loop through each SII category to plot pie charts\nfor sii_value, sii_label in enumerate(sii_labels):\n    # Filter the data for the current SII category\n    sii_data = test[test['sii'] == sii_value]\n    \n    # Count occurrences of each score within the current SII category\n    score_counts = sii_data['sii_square'].value_counts().reindex(score_labels).fillna(0)\n    \n    # Calculate the percentage for each score\n    score_percentages = (score_counts / score_counts.sum()) * 100\n    \n    # Plotting the pie chart for the current SII category\n    plt.figure(figsize=(6, 6))\n    plt.pie(score_percentages, labels=score_labels, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired.colors)\n    \n    # Set the title for each pi\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:07.805404Z","iopub.execute_input":"2024-11-03T23:46:07.805819Z","iopub.status.idle":"2024-11-03T23:46:08.638542Z","shell.execute_reply.started":"2024-11-03T23:46:07.805771Z","shell.execute_reply":"2024-11-03T23:46:08.637063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#question 1,2,3,5\nimport matplotlib.pyplot as plt\n\n# Define the labels for SII categories and PCIAT-PCIAT_Score_square values\nsii_labels = ['None', 'Mild', 'Moderate', 'Severe']\nscore_labels = [0, 1, 2, 3, 4, 5]\n\n# Loop through each SII category to plot pie charts\nfor sii_value, sii_label in enumerate(sii_labels):\n    # Filter the data for the current SII category\n    sii_data = test[test['sii'] == sii_value]\n    \n    # Count occurrences of each score within the current SII category\n    score_counts = sii_data['PCIAT-PCIAT_Score_square'].value_counts().reindex(score_labels).fillna(0)\n    \n    # Calculate the percentage for each score\n    score_percentages = (score_counts / score_counts.sum()) * 100\n    \n    # Plotting the pie chart for the current SII category\n    plt.figure(figsize=(6, 6))\n    plt.pie(score_percentages, labels=score_labels, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired.colors)\n    \n    # Set the title for each pi\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:08.641065Z","iopub.execute_input":"2024-11-03T23:46:08.642146Z","iopub.status.idle":"2024-11-03T23:46:09.603337Z","shell.execute_reply.started":"2024-11-03T23:46:08.642087Z","shell.execute_reply":"2024-11-03T23:46:09.601686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#question 13,18,16\ntest['PCIAT-PCIAT_Score_square_01'] = round(((test['PCIAT-PCIAT_13']+1)*(test['PCIAT-PCIAT_18']+1)*(test['PCIAT-PCIAT_16']+1)) ** (1 / 3))-1","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:09.608556Z","iopub.execute_input":"2024-11-03T23:46:09.609696Z","iopub.status.idle":"2024-11-03T23:46:09.620645Z","shell.execute_reply.started":"2024-11-03T23:46:09.609621Z","shell.execute_reply":"2024-11-03T23:46:09.619248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Define the labels for SII categories and PCIAT-PCIAT_Score_square values\nsii_labels = ['None', 'Mild', 'Moderate', 'Severe']\nscore_labels = [0, 1, 2, 3, 4, 5]\n\n# Loop through each SII category to plot pie charts\nfor sii_value, sii_label in enumerate(sii_labels):\n    # Filter the data for the current SII category\n    sii_data = test[test['sii'] == sii_value]\n    \n    # Count occurrences of each score within the current SII category\n    score_counts = sii_data['PCIAT-PCIAT_Score_square_01'].value_counts().reindex(score_labels).fillna(0)\n    \n    # Calculate the percentage for each score\n    score_percentages = (score_counts / score_counts.sum()) * 100\n    \n    # Plotting the pie chart for the current SII category\n    plt.figure(figsize=(6, 6))\n    plt.pie(score_percentages, labels=score_labels, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired.colors)\n    \n    # Set the title for each pi","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:09.623251Z","iopub.execute_input":"2024-11-03T23:46:09.624295Z","iopub.status.idle":"2024-11-03T23:46:10.431500Z","shell.execute_reply.started":"2024-11-03T23:46:09.624233Z","shell.execute_reply":"2024-11-03T23:46:10.430361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#question 14 20\ntest['PCIAT-PCIAT_Score_square_02'] = round(((test['PCIAT-PCIAT_14']+1)*(test['PCIAT-PCIAT_20']+1)) ** (1 / 2))-1","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:10.433510Z","iopub.execute_input":"2024-11-03T23:46:10.434327Z","iopub.status.idle":"2024-11-03T23:46:10.442811Z","shell.execute_reply.started":"2024-11-03T23:46:10.434272Z","shell.execute_reply":"2024-11-03T23:46:10.441433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Define the labels for SII categories and PCIAT-PCIAT_Score_square values\nsii_labels = ['None', 'Mild', 'Moderate', 'Severe']\nscore_labels = [0, 1, 2, 3, 4, 5]\n\n# Loop through each SII category to plot pie charts\nfor sii_value, sii_label in enumerate(sii_labels):\n    # Filter the data for the current SII category\n    sii_data = test[test['sii'] == sii_value]\n    \n    # Count occurrences of each score within the current SII category\n    score_counts = sii_data['PCIAT-PCIAT_Score_square_02'].value_counts().reindex(score_labels).fillna(0)\n    \n    # Calculate the percentage for each score\n    score_percentages = (score_counts / score_counts.sum()) * 100\n    \n    # Plotting the pie chart for the current SII category\n    plt.figure(figsize=(6, 6))\n    plt.pie(score_percentages, labels=score_labels, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired.colors)\n    \n    # Set the title for each pi","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:10.444898Z","iopub.execute_input":"2024-11-03T23:46:10.445709Z","iopub.status.idle":"2024-11-03T23:46:11.226965Z","shell.execute_reply.started":"2024-11-03T23:46:10.445653Z","shell.execute_reply":"2024-11-03T23:46:11.225092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#question 12 14\ntest['PCIAT-PCIAT_Score_square_03'] = round(((test['PCIAT-PCIAT_12']+1)*(test['PCIAT-PCIAT_04']+1)) ** (1 / 2))-1","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:11.229686Z","iopub.execute_input":"2024-11-03T23:46:11.230295Z","iopub.status.idle":"2024-11-03T23:46:11.241974Z","shell.execute_reply.started":"2024-11-03T23:46:11.230224Z","shell.execute_reply":"2024-11-03T23:46:11.240100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the labels for SII categories and PCIAT-PCIAT_Score_square values\nsii_labels = ['None', 'Mild', 'Moderate', 'Severe']\nscore_labels = [0, 1, 2, 3, 4, 5]\n\n# Loop through each SII category to plot pie charts\nfor sii_value, sii_label in enumerate(sii_labels):\n    # Filter the data for the current SII category\n    sii_data = test[test['sii'] == sii_value]\n    \n    # Count occurrences of each score within the current SII category\n    score_counts = sii_data['PCIAT-PCIAT_Score_square_03'].value_counts().reindex(score_labels).fillna(0)\n    \n    # Calculate the percentage for each score\n    score_percentages = (score_counts / score_counts.sum()) * 100\n    \n    # Plotting the pie chart for the current SII category\n    plt.figure(figsize=(6, 6))\n    plt.pie(score_percentages, labels=score_labels, autopct='%1.1f%%', startangle=140, colors=plt.cm.Paired.colors)\n    \n    # Set the title for each pi","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:11.245259Z","iopub.execute_input":"2024-11-03T23:46:11.245898Z","iopub.status.idle":"2024-11-03T23:46:12.039648Z","shell.execute_reply.started":"2024-11-03T23:46:11.245830Z","shell.execute_reply":"2024-11-03T23:46:12.037794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='sii', y='PCIAT-PCIAT_Score', data=test, palette='Set3')\n\n# Set titles and labels\nplt.title('Distribution of Score across SII Categories')\nplt.xlabel('SII')\nplt.ylabel('Score')\nplt.xticks([0, 1, 2, 3], ['None', 'Mild', 'Moderate', 'Severe'])\n\n# Show plot\nplt.show()\n\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='sii', y='PCIAT-PCIAT_Score_square', data=test, palette='Set3')\n\n# Set titles and labels\nplt.title('Distribution of Score across SII Categories')\nplt.xlabel('SII')\nplt.ylabel('Score')\nplt.xticks([0, 1, 2, 3], ['None', 'Mild', 'Moderate', 'Severe'])\n\n# Show plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:12.042449Z","iopub.execute_input":"2024-11-03T23:46:12.043077Z","iopub.status.idle":"2024-11-03T23:46:12.688482Z","shell.execute_reply.started":"2024-11-03T23:46:12.042991Z","shell.execute_reply":"2024-11-03T23:46:12.687308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count occurrences of each score category by age\nage_score_counts = test.groupby(['Basic_Demos-Age', 'PCIAT-PCIAT_Score_square']).size().unstack(fill_value=0)\n\n# Plotting\nage_score_counts.plot(kind='bar', stacked=True, figsize=(12, 6), colormap=\"viridis\")\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.title('Stacked Bar Chart of Age vs PCIAT-PCIAT_Score_square')\nplt.legend(title='PCIAT-PCIAT_Score_square')\nplt.show()\n\n# Count occurrences of each score category by age\nage_score_counts = test.groupby(['Basic_Demos-Age', 'sii']).size().unstack(fill_value=0)\n\n# Plotting\nage_score_counts.plot(kind='bar', stacked=True, figsize=(12, 6), colormap=\"viridis\")\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.title('Stacked Bar Chart of Age vs sii')\nplt.legend(title='sii')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:12.690050Z","iopub.execute_input":"2024-11-03T23:46:12.690471Z","iopub.status.idle":"2024-11-03T23:46:13.875221Z","shell.execute_reply.started":"2024-11-03T23:46:12.690422Z","shell.execute_reply":"2024-11-03T23:46:13.874063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count occurrences of each score category by age\nage_score_counts = test.groupby(['Basic_Demos-Sex', 'PCIAT-PCIAT_Score_square']).size().unstack(fill_value=0)\n\n# Plotting\nage_score_counts.plot(kind='bar', stacked=True, figsize=(12, 6), colormap=\"viridis\")\nplt.xlabel('Sex')\nplt.ylabel('Count')\nplt.title('Stacked Bar Chart of Age vs PCIAT-PCIAT_Score_square')\nplt.legend(title='PCIAT-PCIAT_Score_square')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:13.878858Z","iopub.execute_input":"2024-11-03T23:46:13.879273Z","iopub.status.idle":"2024-11-03T23:46:14.259767Z","shell.execute_reply.started":"2024-11-03T23:46:13.879232Z","shell.execute_reply":"2024-11-03T23:46:14.258525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import pearsonr\n\n# Create a new dataset with rows containing no NaN values in 'age' or 'PCIAT-PCIAT_Score_square_numeric'\ncleaned_test = test.dropna(subset=['Basic_Demos-Age', 'sii_square'])\n\n# Calculate Pearson correlation if the new dataset is not empty\nif not cleaned_test.empty:\n    from scipy.stats import pearsonr\n    correlation, p_value = pearsonr(cleaned_test['Basic_Demos-Age'], cleaned_test['sii_square'])\n    print(f\"Pearson Correlation: {correlation}, p-value: {p_value}\")\nelse:\n    print(\"No valid data left after removing NaNs.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.261307Z","iopub.execute_input":"2024-11-03T23:46:14.261760Z","iopub.status.idle":"2024-11-03T23:46:14.275689Z","shell.execute_reply.started":"2024-11-03T23:46:14.261689Z","shell.execute_reply":"2024-11-03T23:46:14.274518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import pearsonr\n\ncleaned_test = test.dropna(subset=['Basic_Demos-Age', 'sii'])\n\nif not cleaned_test.empty:\n    correlation, p_value = pearsonr(cleaned_test['Basic_Demos-Age'], cleaned_test['sii'])\n    print(f\"Pearson Correlation: {correlation}, p-value: {p_value}\")\nelse:\n    print(\"No valid data left after removing NaNs.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.276716Z","iopub.execute_input":"2024-11-03T23:46:14.277111Z","iopub.status.idle":"2024-11-03T23:46:14.289284Z","shell.execute_reply.started":"2024-11-03T23:46:14.277052Z","shell.execute_reply":"2024-11-03T23:46:14.287969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import pearsonr\n\ncleaned_test = test.dropna(subset=['PAQ_A-PAQ_A_Total', 'PCIAT-PCIAT_Score_square'])\n\nif not cleaned_test.empty:\n    correlation, p_value = pearsonr(cleaned_test['PAQ_A-PAQ_A_Total'], cleaned_test['PCIAT-PCIAT_Score_square'])\n    print(f\"Pearson Correlation: {correlation}, p-value: {p_value}\")\nelse:\n    print(\"No valid data left after removing NaNs.\")\n    \ncleaned_test = test.dropna(subset=['PAQ_A-PAQ_A_Total', 'sii_square'])\n\nif not cleaned_test.empty:\n    correlation, p_value = pearsonr(cleaned_test['PAQ_A-PAQ_A_Total'], cleaned_test['sii_square'])\n    print(f\"Pearson Correlation: {correlation}, p-value: {p_value}\")\nelse:\n    print(\"No valid data left after removing NaNs.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.290556Z","iopub.execute_input":"2024-11-03T23:46:14.290872Z","iopub.status.idle":"2024-11-03T23:46:14.311711Z","shell.execute_reply.started":"2024-11-03T23:46:14.290836Z","shell.execute_reply":"2024-11-03T23:46:14.310338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import pearsonr\n\ncleaned_test = test.dropna(subset=['FGC-FGC_SRL', 'PCIAT-PCIAT_Score_square'])\n\nif not cleaned_test.empty:\n    correlation, p_value = pearsonr(cleaned_test['FGC-FGC_SRL'], cleaned_test['PCIAT-PCIAT_Score_square'])\n    print(f\"Pearson Correlation: {correlation}, p-value: {p_value}\")\nelse:\n    print(\"No valid data left after removing NaNs.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.313281Z","iopub.execute_input":"2024-11-03T23:46:14.313748Z","iopub.status.idle":"2024-11-03T23:46:14.326530Z","shell.execute_reply.started":"2024-11-03T23:46:14.313695Z","shell.execute_reply":"2024-11-03T23:46:14.325197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport scipy.stats as stats\n\n# Create a contingency table\ncontingency_table = pd.crosstab(test['Basic_Demos-Age'], test['PCIAT-PCIAT_Score_square'])\n\n# Calculate Cramér's V\nchi2 = stats.chi2_contingency(contingency_table)[0]\nn = contingency_table.sum().sum()\nr, k = contingency_table.shape\ncramers_v = np.sqrt(chi2 / (n * (min(r, k) - 1)))\nprint(f\"Cramér's V: {cramers_v}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.327934Z","iopub.execute_input":"2024-11-03T23:46:14.328312Z","iopub.status.idle":"2024-11-03T23:46:14.348195Z","shell.execute_reply.started":"2024-11-03T23:46:14.328272Z","shell.execute_reply":"2024-11-03T23:46:14.347095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count occurrences of each score category by age\nage_score_counts = test.groupby(['Basic_Demos-Age', 'PCIAT-PCIAT_Score_square_01']).size().unstack(fill_value=0)\n\n# Plotting\nage_score_counts.plot(kind='bar', stacked=True, figsize=(12, 6), colormap=\"viridis\")\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.title('Stacked Bar Chart of Age vs PCIAT-PCIAT_Score_square_01')\nplt.legend(title='PCIAT-PCIAT_Score_square')\nplt.show()\n\n# Count occurrences of each score category by age\nage_score_counts = test.groupby(['Basic_Demos-Age', 'sii']).size().unstack(fill_value=0)\n\n# Plotting\nage_score_counts.plot(kind='bar', stacked=True, figsize=(12, 6), colormap=\"viridis\")\nplt.xlabel('Age')\nplt.ylabel('Count')\nplt.title('Stacked Bar Chart of Age vs sii')\nplt.legend(title='sii')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:46:14.349627Z","iopub.execute_input":"2024-11-03T23:46:14.350064Z","iopub.status.idle":"2024-11-03T23:46:15.907056Z","shell.execute_reply.started":"2024-11-03T23:46:14.349987Z","shell.execute_reply":"2024-11-03T23:46:15.905928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Create a new DataFrame with combinations as rows\ncombination_counts = df[columns].astype(str).agg(' '.join, axis=1).value_counts()\n\n# Convert to a DataFrame\ncombination_counts_df = combination_counts.reset_index()\ncombination_counts_df.columns = ['Combination', 'Count']\n\n# Calculate the average count\naverage_count = combination_counts_df['Count'].mean()\n# Filter combinations with count greater than the average\nfiltered_combinations_df = combination_counts_df[combination_counts_df['Count'] > average_count]\n\n# Display the filtered result\nprint(filtered_combinations_df)\n\n# Visualization\nplt.figure(figsize=(10, 6))\nplt.bar(filtered_combinations_df['Combination'], filtered_combinations_df['Count'], color='skyblue')\nplt.xlabel('Score Combinations')\nplt.ylabel('Count')\nplt.title('Count of Score Combinations Above Average')\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()  # Adjust layout to make room for x-axis labels\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:47:55.103568Z","iopub.execute_input":"2024-11-03T23:47:55.104000Z","iopub.status.idle":"2024-11-03T23:47:55.531129Z","shell.execute_reply.started":"2024-11-03T23:47:55.103958Z","shell.execute_reply":"2024-11-03T23:47:55.530064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Assuming 'test' is your DataFrame\ndf = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Create a new DataFrame with combinations as rows\ncombination_counts = df.astype(str).agg(' '.join, axis=1).value_counts()\n\n# Convert to a DataFrame\ncombination_counts_df = combination_counts.reset_index()\ncombination_counts_df.columns = ['Combination', 'Count']\n\n# Count combinations that are \"almost\" all 3s\nalmost_all_3_counts = 0\nalmost_all_3_combinations = []\n\n# Iterate through each combination and count the ones that are almost all 3s\nfor index, row in combination_counts_df.iterrows():\n    combination = row['Combination']\n    count = row['Count']\n    \n    # Convert values to floats and then to integers\n    values = [int(float(v)) for v in combination.split()]\n    non_three_count = sum(1 for v in values if v != 0)\n    \n    if non_three_count <= 2:  # Less than or equal to 2 values are not 3\n        almost_all_3_counts += count\n        almost_all_3_combinations.append((combination, count))\n\n# Display the results\nprint(\"Counts of Combinations That Are 'Almost' All 3s:\")\nfor comb, count in almost_all_3_combinations:\n    print(f\"Combination: {comb}, Count: {count}\")\n\nprint(f\"\\nTotal Count of 'Almost' All 3s: {almost_all_3_counts}\")\n\n# Visualization (if needed)\n# Prepare DataFrame for visualization of \"almost\" all 3s\nalmost_all_3_df = pd.DataFrame(almost_all_3_combinations, columns=['Combination', 'Count'])\n\nplt.figure(figsize=(10, 6))\nplt.bar(almost_all_3_df['Combination'], almost_all_3_df['Count'], color='lightgreen')\nplt.xlabel('Score Combinations')\nplt.ylabel('Count')\nplt.title('Count of Combinations That Are \"Almost\" All 3s')\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()  # Adjust layout to make room for x-axis labels\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T23:56:01.417154Z","iopub.execute_input":"2024-11-03T23:56:01.417569Z","iopub.status.idle":"2024-11-03T23:56:01.947793Z","shell.execute_reply.started":"2024-11-03T23:56:01.417530Z","shell.execute_reply":"2024-11-03T23:56:01.946699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming 'test' is your DataFrame\ndf = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Convert values to integers\ndf[columns] = df[columns].astype(int)\n\n# Function to check for sequential numbers\ndef is_sequential(values):\n    return values == list(range(min(values), max(values) + 1))\n\n# Function to check for repeated patterns\ndef has_repeated_pattern(values):\n    # Convert to string\n    value_str = ''.join(map(str, values))\n    # Check for repeating patterns in the string\n    for length in range(1, len(value_str) // 2 + 1):\n        if value_str[:length] * (len(value_str) // length) == value_str:\n            return True\n    return False\n\n# Check for any tester who filled in a specific pattern\npattern_found = []\nfor index, row in df.iterrows():\n    values = row.tolist()\n\n    # Check for various patterns\n    if (is_sequential(values) or \n        has_repeated_pattern(values) or\n        all(v == values[0] for v in values)):  # All values the same\n        pattern_found.append((index, values))\n\n# Display the results\nif pattern_found:\n    print(\"Testers who filled in the form with specific patterns:\")\n    for idx, val in pattern_found:\n        print(f\"Tester Index: {idx}, Responses: {val}\")\nelse:\n    print(\"No testers filled in the form with the specified patterns.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T00:00:12.337512Z","iopub.execute_input":"2024-11-04T00:00:12.337955Z","iopub.status.idle":"2024-11-04T00:00:12.390422Z","shell.execute_reply.started":"2024-11-04T00:00:12.337906Z","shell.execute_reply":"2024-11-04T00:00:12.389351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming 'test' is your DataFrame\ndf = test\n\n# Define the columns to consider\ncolumns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\n\n# Drop rows with any NaN values in the specified columns\ndf = df[columns].dropna()\n\n# Convert values to integers\ndf[columns] = df[columns].astype(int)\n\n# Function to check for symmetry\ndef is_symmetric(values):\n    values_str = ''.join(map(str, values))\n    return values_str == values_str[::-1]\n\n# Check for any tester who filled in a symmetrical pattern\nsymmetric_patterns_found = []\nfor index, row in df.iterrows():\n    values = row.tolist()\n    if is_symmetric(values):\n        symmetric_patterns_found.append((index, values))\n\n# Display the results\nif symmetric_patterns_found:\n    print(\"Testers who filled in the form with symmetrical patterns:\")\n    for idx, val in symmetric_patterns_found:\n        print(f\"Tester Index: {idx}, Responses: {val}\")\nelse:\n    print(\"No testers filled in the form with symmetrical patterns.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T00:00:17.884123Z","iopub.execute_input":"2024-11-04T00:00:17.884568Z","iopub.status.idle":"2024-11-04T00:00:17.929204Z","shell.execute_reply.started":"2024-11-04T00:00:17.884511Z","shell.execute_reply":"2024-11-04T00:00:17.928032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}