{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Hiểu dữ liệu:","metadata":{}},{"cell_type":"markdown","source":"## Mục tiêu và bối cảnh dữ liệu","metadata":{}},{"cell_type":"markdown","source":"👉 HBN Dataset: Mẫu nghiên cứu lâm sàng từ **5.000** trẻ em và thanh niên (5–22 tuổi) nhằm xác định các chỉ dấu sinh học hỗ trợ chẩn đoán và điều trị các rối loạn sức khỏe tâm thần và học tập.\n\n👉 Dữ liệu sử dụng trong cuộc thi:\n- Dữ liệu hoạt động thể chất: Dữ liệu từ máy đo gia tốc đeo tay, đánh giá thể chất, và bảng câu hỏi.\n- Dữ liệu hành vi sử dụng Internet.\n\n👉 Mục tiêu cuộc thi: Dự đoán **Severity Impairment Index (sii)** - thước đo mức độ sử dụng Internet có vấn đề, từ các tập dữ liệu này.","metadata":{}},{"cell_type":"markdown","source":"## Thông tin về tập dữ liệu","metadata":{}},{"cell_type":"markdown","source":"👉 Dạng dữ liệu:\n- Dữ liệu actigraphy (gia tốc kế) trong file parquet.\n- Dữ liệu dạng bảng trong file CSV.\n\n👉 Đặc điểm dữ liệu:\n- Nhiều giá trị bị thiếu, đặc biệt là trong tập huấn luyện.\n  \n👉 Khoảng giá trị sii:\n- Dựa trên trường PCIAT_Total (Parent-Child Internet Addiction Test).\n- Phân loại mức độ: 0 (None), 1 (Mild), 2 (Moderate), 3 (Severe).","metadata":{}},{"cell_type":"markdown","source":"\n\n# 2. Xem xét dữ liệu:","metadata":{},"attachments":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport warnings\nimport polars as pl\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\ntarget_labels = ['None', 'Mild', 'Moderate', 'Severe']\n\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\nsns.set(style=\"whitegrid\")\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.705309Z","iopub.execute_input":"2024-12-22T04:38:12.705637Z","iopub.status.idle":"2024-12-22T04:38:12.712729Z","shell.execute_reply.started":"2024-12-22T04:38:12.705612Z","shell.execute_reply":"2024-12-22T04:38:12.711116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.727285Z","iopub.execute_input":"2024-12-22T04:38:12.727628Z","iopub.status.idle":"2024-12-22T04:38:12.773894Z","shell.execute_reply.started":"2024-12-22T04:38:12.727597Z","shell.execute_reply":"2024-12-22T04:38:12.773080Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train data","metadata":{}},{"cell_type":"code","source":"display(train.head())\nprint(f\"Train shape: {train.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.792706Z","iopub.execute_input":"2024-12-22T04:38:12.792976Z","iopub.status.idle":"2024-12-22T04:38:12.825502Z","shell.execute_reply.started":"2024-12-22T04:38:12.792945Z","shell.execute_reply":"2024-12-22T04:38:12.824557Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test data","metadata":{}},{"cell_type":"code","source":"display(test.head())\nprint(f\"Test shape: {test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.868848Z","iopub.execute_input":"2024-12-22T04:38:12.869084Z","iopub.status.idle":"2024-12-22T04:38:12.890251Z","shell.execute_reply.started":"2024-12-22T04:38:12.869064Z","shell.execute_reply":"2024-12-22T04:38:12.889283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(data_dict.head())\nprint(f\"Dictionary shape: {data_dict.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.900866Z","iopub.execute_input":"2024-12-22T04:38:12.901123Z","iopub.status.idle":"2024-12-22T04:38:12.915619Z","shell.execute_reply.started":"2024-12-22T04:38:12.901102Z","shell.execute_reply":"2024-12-22T04:38:12.914734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing value","metadata":{}},{"cell_type":"code","source":"season_dtype = pl.Enum(['Spring', 'Summer', 'Fall', 'Winter'])\n\ntrain = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntest = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\nmissing_count = (\n    train\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(train)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title('Missing values over the whole training dataset')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:12.916747Z","iopub.execute_input":"2024-12-22T04:38:12.916959Z","iopub.status.idle":"2024-12-22T04:38:14.074259Z","shell.execute_reply.started":"2024-12-22T04:38:12.916941Z","shell.execute_reply":"2024-12-22T04:38:14.073298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cột nào cũng có dữ liệu bị mất trừ cột id (tất nhiên), sex, age và season of enrollment, kể cả cột sii","metadata":{}},{"cell_type":"code","source":"supervised_usable = (\n    train\n    .filter(pl.col('sii').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:14.075596Z","iopub.execute_input":"2024-12-22T04:38:14.075842Z","iopub.status.idle":"2024-12-22T04:38:15.287198Z","shell.execute_reply.started":"2024-12-22T04:38:14.075820Z","shell.execute_reply":"2024-12-22T04:38:15.286280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nếu chúng ta chỉ tính missing values cho những dữ liệu train mà có target variable đã biết thì biểu đồ sẽ khác một chúc","metadata":{}},{"cell_type":"markdown","source":"# 3. Phân tích dữ liệu định lượng:","metadata":{}},{"cell_type":"markdown","source":"## Biến `sii`","metadata":{}},{"cell_type":"code","source":"# print(train.select(pl.col('PCIAT-PCIAT_Total').is_null() == pl.col('sii').is_null()).to_series().mean())\n\n(train\n .select(pl.col('PCIAT-PCIAT_Total'))\n .group_by(train.get_column('sii'))\n .agg(pl.col('PCIAT-PCIAT_Total').min().alias('PCIAT-PCIAT_Total min'),\n      pl.col('PCIAT-PCIAT_Total').max().alias('PCIAT-PCIAT_Total max'),\n      pl.col('PCIAT-PCIAT_Total').len().alias('count'))\n .sort('sii')\n)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:15.288289Z","iopub.execute_input":"2024-12-22T04:38:15.288520Z","iopub.status.idle":"2024-12-22T04:38:15.296175Z","shell.execute_reply.started":"2024-12-22T04:38:15.288498Z","shell.execute_reply":"2024-12-22T04:38:15.295232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tập dữ liệu kiểm tra không có cột PCIAT\n\n💡 Insight:\n\n- Chúng ta nên tập trung vào dự đoán mục tiêu từ tất cả các đặc trưng khác ngoại trừ kết quả PCIAT.\n- Chúng ta chỉ biết mục tiêu đối với hai phần ba mẫu dữ liệu. Các mẫu không có mục tiêu có thể được sử dụng cho học bán giám sát.\n- Chúng ta có thể trực tiếp dự đoán sii (đây là giá trị mục tiêu), hoặc chúng ta có thể dự đoán PCIAT-PCIAT_Total rồi chuyển đổi dự đoán này thành dự đoán `sii` để gửi. Vì PCIAT-PCIAT_Total chi tiết và cung cấp thông tin hơn `sii`, nên việc huấn luyện để dự đoán PCIAT-PCIAT_Total có tiềm năng tạo ra một mô hình tốt hơn.","metadata":{}},{"cell_type":"markdown","source":"## Boy and girl (Nam và nữ)\nNgười tham gia nghiên cứu có độ tuổi từ 5 đến 22 tuổi. Số lượng con trai gấp đôi con gái","metadata":{}},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('Basic_Demos-Age').value_counts()\n    ax.bar(vc.get_column('Basic_Demos-Age'),\n           vc.get_column('count'),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.xaxis.set_major_locator(MaxNLocator(integer=True))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Age distribution')\naxs.ravel()[1].set_xlabel('years')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:15.297214Z","iopub.execute_input":"2024-12-22T04:38:15.297451Z","iopub.status.idle":"2024-12-22T04:38:15.827771Z","shell.execute_reply.started":"2024-12-22T04:38:15.297432Z","shell.execute_reply":"2024-12-22T04:38:15.826867Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nam có nguy cơ gặp vấn đề sử dụng quá độ nhiều hơn nữ","metadata":{}},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True, sharey=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('sii').value_counts()\n    ax.bar(vc.get_column('sii'),\n           vc.get_column('count') / vc.get_column('count').sum(),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.set_xticks(np.arange(4), target_labels)\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Target distribution')\naxs.ravel()[1].set_xlabel('Severity Impairment Index (sii)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:15.828771Z","iopub.execute_input":"2024-12-22T04:38:15.829131Z","iopub.status.idle":"2024-12-22T04:38:16.165120Z","shell.execute_reply.started":"2024-12-22T04:38:15.829106Z","shell.execute_reply":"2024-12-22T04:38:16.164093Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep (Giấc ngủ của người tham gia)","metadata":{}},{"cell_type":"code","source":"vc = train.get_column('Physical-HeartRate').value_counts()\ncolor = np.where(vc.get_column('Physical-HeartRate') < 50, 'r', 'b')\nplt.figure(figsize=(8, 2))\nplt.title('Biểu đồ phân bố Physical-HeartRate, với các outlier ở đầu cuối')\nplt.bar(vc.get_column('Physical-HeartRate'), vc.get_column('count'), color=color)\nplt.xlabel('Physical-HeartRate')\nplt.ylabel('count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:16.167720Z","iopub.execute_input":"2024-12-22T04:38:16.167946Z","iopub.status.idle":"2024-12-22T04:38:16.551485Z","shell.execute_reply.started":"2024-12-22T04:38:16.167927Z","shell.execute_reply":"2024-12-22T04:38:16.550491Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Câu hỏi về thang đo rối loạn giấc ngủ 🛌 cung cấp một điểm số thô từ 0 đến 100. Điểm số thô sau đó được chuyển đổi thành t-score.C  ó thể bỏ một trong các đặc trưng mà không mất thông tin.\n\nPhân bố giấc ngủ 💤 t-score được định nghĩa sao cho trung bình là 50 và độ lệch chuẩn là 10. Rõ ràng, chúng ta có 29 trẻ em có giấc ngủ kém hơn trung bình của dân số chung đến năm độ lệch chuẩn. Điều này là không hợp lý lắm? 🤔","metadata":{}},{"cell_type":"code","source":"vc = train.get_column('SDS-SDS_Total_Raw').value_counts()\nplt.figure(figsize=(6, 2))\nplt.title('Sleep disturbance scale')\nplt.bar(vc.get_column('SDS-SDS_Total_Raw'), vc.get_column('count'), color='brown')\nplt.xlabel('SDS-SDS_Total_Raw')\nplt.ylabel('count')\nplt.show()\n\nplt.title('Sleep disturbance scale: conversion from raw to t score')\nplt.scatter(train.get_column('SDS-SDS_Total_Raw'),\n            train.get_column('SDS-SDS_Total_T'),\n            color='brown')\nplt.xlabel('SDS-SDS_Total_Raw')\nplt.ylabel('SDS-SDS_Total_T')\nplt.show()\n\nvc = train.get_column('SDS-SDS_Total_T').value_counts()\nplt.figure(figsize=(6, 2))\nplt.title('Sleep disturbance scale')\nplt.bar(vc.get_column('SDS-SDS_Total_T'), vc.get_column('count'), color='brown')\nplt.xlabel('SDS-SDS_Total_T')\nplt.ylabel('count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:16.553121Z","iopub.execute_input":"2024-12-22T04:38:16.553391Z","iopub.status.idle":"2024-12-22T04:38:17.547905Z","shell.execute_reply.started":"2024-12-22T04:38:16.553369Z","shell.execute_reply":"2024-12-22T04:38:17.546989Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"💡 **Insight:** Có nhiều outliers mà chúng ta cần xử lý","metadata":{}},{"cell_type":"markdown","source":"**💡 Insight**: <ul style=\"list-style:circle\"> <li>Cả điểm thô và điểm T về rối loạn giấc ngủ đều có mức độ biến động trung bình, với một số giá trị cực đoan cho thấy tình trạng rối loạn giấc ngủ nghiêm trọng ở một nhóm nhỏ người tham gia. </ul> </div>","metadata":{}},{"cell_type":"markdown","source":"## Time-series data (Các tệp Actigraphy)\n**Actigraphy** (đeo máy đo hoạt động) là một phương pháp không xâm lấn để theo dõi chu kỳ nghỉ ngơi/hoạt động của con người. Một thiết bị đeo nhỏ gọn, còn được gọi là cảm biến đo hoạt động, được đeo trong một tuần hoặc hơn để đo hoạt động vận động cơ lớn.\n\nThiết bị này thường được đeo ở cổ tay giống như đồng hồ đeo tay. Các chuyển động của thiết bị đo hoạt động được ghi lại liên tục và một số thiết bị còn đo cả lượng ánh sáng tiếp xúc.\n\nChúng ta có các tệp actigraphy cho một phần tư số người tham gia (chính xác là 996 người). Tên tệp luôn là **part-0.parquet**.\n\nXem xét hồ sơ của người tham gia id=0417c91e, một bé gái thuận tay phải sáu tuổi, chúng ta thấy rằng người tham gia này bắt đầu sử dụng máy đo gia tốc vào thứ Ba (ngày trong tuần = 2) của quý thứ hai trong năm, vào giây thứ 44100 của ngày (12:15 CH), 5 ngày sau kỳ thi PCIAT. Cô bé trả lại máy đo gia tốc vào ngày thứ 53 sau kỳ thi PCIAT, vào thứ Hai của quý thứ ba, lúc 9:08 SA.\n\nTrang dữ liệu cuộc thi cho biết `time_of_day` có định dạng `%H:%M:%S.%9f`. Rõ ràng đây là không chính xác. `time_of_day` được đo bằng nano giây kể từ nửa đêm.","metadata":{}},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.548844Z","iopub.execute_input":"2024-12-22T04:38:17.549196Z","iopub.status.idle":"2024-12-22T04:38:17.582169Z","shell.execute_reply.started":"2024-12-22T04:38:17.549172Z","shell.execute_reply":"2024-12-22T04:38:17.581294Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Chúng ta có thể vẽ sơ đồ chuỗi thời gian trong tệp này. Chúng ta có thể thấy gì?\n1. Chúng ta thấy rõ mô hình hàng ngày.\n2. Chúng ta thấy cô gái đeo máy đo gia tốc trong 31 ngày rồi tháo nó ra.\n3. Tập dữ liệu có cột non-wear_flag nhưng cờ đó luôn bằng 0 đối với người tham gia này. \n4. Cô gái ở trong môi trường có độ chiếu sáng vượt quá 2500 [lux](https://en.wikipedia.org/wiki/Lux) mỗi ngày (thiết bị không thể đo quá 2500 lux). Độ chiếu sáng cao như vậy có nghĩa là cô ấy đang ở ngoài trời hoặc trong một căn phòng có cửa sổ lớn.\n5. Cô gái di chuyển rất nhiều: cô ấy có giá trị enmo trên 2 hầu như mỗi ngày.\n6. Chuỗi thời gian thường chứa các phép đo cứ sau 5 giây, nhưng thiếu một số bước thời gian. Nó không được ghi lại trong những điều kiện nào mà các bước thời gian bị bỏ qua.\n\n💡 **Insight:** \n- Sử dụng ENMO và cột ánh sáng và đừng tin vào cờ non-wear!\n- ENMO và các cột ánh sáng tự cung cấp để phân tích bằng mạng nơ-ron tích chập một chiều, nhưng nếu muốn bắt đầu đơn giản, chúng ta có thể sử dụng một số tập hợp cơ bản (trung bình, phương sai, ...) của chuỗi thời gian làm đặc điểm cho một mô hình tăng cường độ dốc.","metadata":{}},{"cell_type":"markdown","source":"## Age (Tuổi)","metadata":{}},{"cell_type":"code","source":"groups = data_dict.groupby('Instrument')['Field'].apply(list).to_dict()\nfor instrument, features in groups.items():\n    print(f\"{instrument}: {features}\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.583155Z","iopub.execute_input":"2024-12-22T04:38:17.583470Z","iopub.status.idle":"2024-12-22T04:38:17.594622Z","shell.execute_reply.started":"2024-12-22T04:38:17.583439Z","shell.execute_reply":"2024-12-22T04:38:17.593703Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.595496Z","iopub.execute_input":"2024-12-22T04:38:17.595900Z","iopub.status.idle":"2024-12-22T04:38:17.615827Z","shell.execute_reply.started":"2024-12-22T04:38:17.595868Z","shell.execute_reply":"2024-12-22T04:38:17.614895Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.616716Z","iopub.execute_input":"2024-12-22T04:38:17.617014Z","iopub.status.idle":"2024-12-22T04:38:17.668095Z","shell.execute_reply.started":"2024-12-22T04:38:17.616968Z","shell.execute_reply":"2024-12-22T04:38:17.667139Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_columns = [col for col in train.columns if 'Season' in col]\nseason_df = train[season_columns]","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.669064Z","iopub.execute_input":"2024-12-22T04:38:17.669288Z","iopub.status.idle":"2024-12-22T04:38:17.674697Z","shell.execute_reply.started":"2024-12-22T04:38:17.669269Z","shell.execute_reply":"2024-12-22T04:38:17.673627Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[season_columns] = train[season_columns].fillna(\"Missing\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.675603Z","iopub.execute_input":"2024-12-22T04:38:17.675877Z","iopub.status.idle":"2024-12-22T04:38:17.701572Z","shell.execute_reply.started":"2024-12-22T04:38:17.675850Z","shell.execute_reply":"2024-12-22T04:38:17.700770Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict = data_dict[data_dict['Instrument'] != 'Parent-Child Internet Addiction Test']\ncontinuous_cols = data_dict[data_dict['Type'].str.contains(\n    'float|int', case=False\n)]['Field'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.702519Z","iopub.execute_input":"2024-12-22T04:38:17.702849Z","iopub.status.idle":"2024-12-22T04:38:17.717908Z","shell.execute_reply.started":"2024-12-22T04:38:17.702818Z","shell.execute_reply":"2024-12-22T04:38:17.716944Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert train['Basic_Demos-Age'].isna().sum() == 0\nassert train['Basic_Demos-Sex'].isna().sum() == 0","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.718801Z","iopub.execute_input":"2024-12-22T04:38:17.719164Z","iopub.status.idle":"2024-12-22T04:38:17.734979Z","shell.execute_reply.started":"2024-12-22T04:38:17.719131Z","shell.execute_reply":"2024-12-22T04:38:17.734222Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Age Group'] = pd.cut(\n    train['Basic_Demos-Age'],\n    bins=[4, 12, 18, 22],\n    labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)\ncalculate_stats(train, 'Age Group')","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.735954Z","iopub.execute_input":"2024-12-22T04:38:17.736251Z","iopub.status.idle":"2024-12-22T04:38:17.759027Z","shell.execute_reply.started":"2024-12-22T04:38:17.736225Z","shell.execute_reply":"2024-12-22T04:38:17.758103Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sex_map = {0: 'Male', 1: 'Female'}\ntrain['Basic_Demos-Sex'] = train['Basic_Demos-Sex'].map(sex_map)\ncalculate_stats(train, 'Basic_Demos-Sex')","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.759927Z","iopub.execute_input":"2024-12-22T04:38:17.760177Z","iopub.status.idle":"2024-12-22T04:38:17.780199Z","shell.execute_reply.started":"2024-12-22T04:38:17.760157Z","shell.execute_reply":"2024-12-22T04:38:17.779219Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"),\n        labeldistance=1.05, pctdistance=0.80\n    )\n    axes[i].set_title(f'SII Distribution for {age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:17.787084Z","iopub.execute_input":"2024-12-22T04:38:17.787387Z","iopub.status.idle":"2024-12-22T04:38:18.288422Z","shell.execute_reply.started":"2024-12-22T04:38:17.787358Z","shell.execute_reply":"2024-12-22T04:38:18.287634Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.290990Z","iopub.execute_input":"2024-12-22T04:38:18.291364Z","iopub.status.idle":"2024-12-22T04:38:18.309047Z","shell.execute_reply.started":"2024-12-22T04:38:18.291334Z","shell.execute_reply":"2024-12-22T04:38:18.308072Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Calculate percentages for participants with non-missing SII only:","metadata":{}},{"cell_type":"code","source":"stats = train[train['sii'] != 'Missing'].groupby(\n    ['Age Group', 'sii']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.310119Z","iopub.execute_input":"2024-12-22T04:38:18.310411Z","iopub.status.idle":"2024-12-22T04:38:18.328900Z","shell.execute_reply.started":"2024-12-22T04:38:18.310381Z","shell.execute_reply":"2024-12-22T04:38:18.327974Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**💡 Insight**: <ul style=\"list-style:circle\"> <li>Biểu đồ hộp là các cách biểu diễn khác nhau của biến mục tiêu ở dạng phân loại (SII) và dạng số (PCIAT_Total). Chúng cho thấy rằng điểm SII cao hơn thường liên quan đến các nhóm tuổi lớn hơn, nhưng có sự chồng lấn đáng kể trong các khoảng tuổi của mỗi nhóm, và trung vị PCIAT_Total cao hơn ở thanh thiếu niên, gợi ý một mối quan hệ hình chữ U giữa tuổi và suy giảm PIU (đỉnh điểm của các vấn đề liên quan đến Internet có thể xảy ra trong giai đoạn thanh thiếu niên). <li>Phù hợp với điều đó, trong các biểu đồ tròn, phân phối SII ở trẻ em và người lớn nghiêng về các giá trị thấp hơn (không có hoặc nhẹ), trong khi ở thanh thiếu niên, phân phối cân bằng hơn giữa các danh mục không có, nhẹ và trung bình. <li>Nhưng còn về số lượng thì sao (xem bảng)? Số lượng thanh thiếu niên ít hơn nhiều so với trẻ em, và số lượng người lớn tham gia cực kỳ thấp (tổng cộng 88 và chỉ 36 người có SII)! <li>Như chúng ta đã thấy từ các biểu đồ trong phần trước, phân phối tổng thể của SII nghiêng về các giá trị thấp hơn và các trường hợp nghiêm trọng rất hiếm. Vì vậy, có thể có những mối quan hệ mà chúng ta không thể thấy do kích thước mẫu không đồng đều và sự thiếu đại diện của các trường hợp nghiêm trọng. <li>Sự khác biệt giữa nam và nữ tương đối nhỏ. </ul> </div>","metadata":{}},{"cell_type":"markdown","source":"## Internet Usage (Mức độ sử dụng Internet)","metadata":{}},{"cell_type":"markdown","source":"Dữ liệu sử dụng Internet đóng vai trò quan trọng trong nhiệm vụ này vì việc sử dụng Internet có vấn đề (PIU), còn được gọi là nghiện Internet hoặc sử dụng Internet một cách cưỡng chế, đề cập đến việc sử dụng Internet quá mức và không lành mạnh, làm gián đoạn cuộc sống hàng ngày, trách nhiệm và các mối quan hệ xã hội của một người. Dữ liệu sử dụng Internet cung cấp một thước đo trực tiếp về thời gian mỗi người tham gia dành cho việc trực tuyến.","metadata":{}},{"cell_type":"code","source":"data = train[train['PreInt_EduHx-computerinternet_hoursday'].notna()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Khoảng tuổi của những người tham gia có dữ liệu được đo về PreInt_EduHx-computerinternet_hoursday:\"\n    f\" {age_range.min()} - {age_range.max()} tuổi\"\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.329909Z","iopub.execute_input":"2024-12-22T04:38:18.330242Z","iopub.status.idle":"2024-12-22T04:38:18.337953Z","shell.execute_reply.started":"2024-12-22T04:38:18.330211Z","shell.execute_reply":"2024-12-22T04:38:18.337114Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['PreInt_EduHx-computerinternet_hoursday'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.338583Z","iopub.execute_input":"2024-12-22T04:38:18.338893Z","iopub.status.idle":"2024-12-22T04:38:18.358618Z","shell.execute_reply.started":"2024-12-22T04:38:18.338866Z","shell.execute_reply":"2024-12-22T04:38:18.357717Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_map = {0: '< 1h/day', 1: '~ 1h/day', 2: '~ 2hs/day', 3: '> 3hs/day'}\ntrain['internet_use_encoded'] = train[\n    'PreInt_EduHx-computerinternet_hoursday'\n].map(param_map).fillna('Missing')\n\nparam_ord = ['Missing', '< 1h/day', '~ 1h/day', '~ 2hs/day', '> 3hs/day']\ntrain['internet_use_encoded'] = pd.Categorical(\n    train['internet_use_encoded'], categories=param_ord,\n    ordered=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.359609Z","iopub.execute_input":"2024-12-22T04:38:18.359883Z","iopub.status.idle":"2024-12-22T04:38:18.378899Z","shell.execute_reply.started":"2024-12-22T04:38:18.359856Z","shell.execute_reply":"2024-12-22T04:38:18.378101Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# Hours of Internet Use\nax1 = sns.countplot(x='internet_use_encoded', data=train, palette=\"Set3\", ax=axes[0])\naxes[0].set_title('Distribution of Hours of Internet Use')\naxes[0].set_xlabel('Hours per Day Group')\naxes[0].set_ylabel('Count')\n\ntotal = len(train['internet_use_encoded'])\nfor p in ax1.patches:\n    count = int(p.get_height())\n    percentage = '{:.1f}%'.format(100 * count / total)\n    ax1.annotate(f'{count} ({percentage})', (p.get_x() + p.get_width() / 2., p.get_height()), \n                 ha='center', va='baseline', fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')\n\n# Hours of Internet Use by Age\nsns.boxplot(y=train['Basic_Demos-Age'], x=train['internet_use_encoded'], ax=axes[1], palette=\"Set3\")\naxes[1].set_title('Hours of Internet Use by Age')\naxes[1].set_ylabel('Age')\naxes[1].set_xlabel('Hours per Day Group')\n\n# Hours of Internet Use (numeric) by Age Group\nsns.boxplot(y='PreInt_EduHx-computerinternet_hoursday', x='Age Group', data=train, ax=axes[2], palette=\"Set3\")\naxes[2].set_title('Internet Hours by Age Group')\naxes[2].set_ylabel('Hours per Day (Numeric)')\naxes[2].set_xlabel('Age Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:18.379804Z","iopub.execute_input":"2024-12-22T04:38:18.380090Z","iopub.status.idle":"2024-12-22T04:38:19.203349Z","shell.execute_reply.started":"2024-12-22T04:38:18.380062Z","shell.execute_reply":"2024-12-22T04:38:19.202288Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(\n    ['Age Group', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(group_counts, labels=group_counts.index, autopct='%1.1f%%',\n                startangle=90, colors=sns.color_palette(\"Set3\"), labeldistance=1.1)\n    axes[i].set_title(f'Distribution of Hours of Internet Use\\n{age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.204368Z","iopub.execute_input":"2024-12-22T04:38:19.204654Z","iopub.status.idle":"2024-12-22T04:38:19.709228Z","shell.execute_reply.started":"2024-12-22T04:38:19.204625Z","shell.execute_reply":"2024-12-22T04:38:19.708240Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_non_na = train.dropna(subset=['PreInt_EduHx-computerinternet_hoursday'])\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday'] == 3).sum()\nprint(f\"Non-NA Rows - Internet use 3h or more: {(rows / len(train_non_na)) * 100:.2f}%\")\n\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday'] == 0).sum()\nprint(f\"Non-NA Rows - Internet use 1h or less: {(rows / len(train_non_na)) * 100:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.710345Z","iopub.execute_input":"2024-12-22T04:38:19.710669Z","iopub.status.idle":"2024-12-22T04:38:19.722604Z","shell.execute_reply.started":"2024-12-22T04:38:19.710639Z","shell.execute_reply":"2024-12-22T04:38:19.721711Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Basic_Demos-Sex', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.723837Z","iopub.execute_input":"2024-12-22T04:38:19.724201Z","iopub.status.idle":"2024-12-22T04:38:19.757908Z","shell.execute_reply.started":"2024-12-22T04:38:19.724169Z","shell.execute_reply":"2024-12-22T04:38:19.756927Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"💡 Lưu ý: <ul style=\"list-style:circle\"> <li>Dữ liệu về sử dụng Internet bị thiếu ở 16,6% số người tham gia, trong khi 38,5% báo cáo sử dụng Internet dưới 1 giờ mỗi ngày. <li>Tương tự như dữ liệu SII, biểu đồ hộp cho thấy rằng việc sử dụng Internet hàng ngày cao hơn có liên quan đến nhóm tuổi lớn hơn, nhưng có sự chồng lấn đáng kể trong các khoảng tuổi của mỗi danh mục sử dụng Internet. Tuy nhiên, cả dạng phân loại và dạng số của số giờ trực tuyến đều chỉ ra một mối quan hệ tuyến tính nhất quán. <li>Biểu đồ tròn theo nhóm tuổi phù hợp và thể hiện điều tương tự. <li>Tạo một đặc trưng tương tác giữa việc sử dụng Internet và tuổi có thể hữu ích cho việc xây dựng mô hình. <li>Việc sử dụng Internet tương đối giống nhau ở cả hai giới. </ul> </div>\n\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# 4. Tương quan giữa các biến:","metadata":{}},{"cell_type":"markdown","source":"## Weight và Height","metadata":{}},{"cell_type":"code","source":"wh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.758860Z","iopub.execute_input":"2024-12-22T04:38:19.759118Z","iopub.status.idle":"2024-12-22T04:38:19.762933Z","shell.execute_reply.started":"2024-12-22T04:38:19.759098Z","shell.execute_reply":"2024-12-22T04:38:19.761567Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\nGiá trị tối thiểu là 0 đối với các chỉ số như BMI, cân nặng và huyết áp là không thực tế về mặt sinh học và có khả năng cho thấy dữ liệu bị thiếu hoặc sai lệch. Hãy kiểm tra số lượng giá trị bằng 0 trong các cột này:","metadata":{}},{"cell_type":"code","source":"(train[wh_cols] == 0).sum()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.764180Z","iopub.execute_input":"2024-12-22T04:38:19.764494Z","iopub.status.idle":"2024-12-22T04:38:19.781301Z","shell.execute_reply.started":"2024-12-22T04:38:19.764464Z","shell.execute_reply":"2024-12-22T04:38:19.780546Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Thay thế các giá trị 0 bằng NaN và kiểm tra lại các thống kê:","metadata":{}},{"cell_type":"code","source":"train[wh_cols] = train[wh_cols].replace(0, np.nan)\ncalculate_stats(train, wh_cols)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.782268Z","iopub.execute_input":"2024-12-22T04:38:19.782557Z","iopub.status.idle":"2024-12-22T04:38:19.815354Z","shell.execute_reply.started":"2024-12-22T04:38:19.782530Z","shell.execute_reply":"2024-12-22T04:38:19.814533Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Chuyển trọng lượng sang kilôgam và chiều cao sang xentimét, sau đó tính lại chỉ số BMI:","metadata":{}},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ntrain['Physical-Weight'] = train['Physical-Weight'] * lbs_to_kg\ntrain['Physical-Height'] = train['Physical-Height'] * inches_to_cm\ntrain['Physical-Waist_Circumference'] = train['Physical-Waist_Circumference'] * inches_to_cm\n\n# Recalculate BMI: BMI = weight (kg) / (height (m)^2)\ntrain['Physical-BMI'] = np.where(\n    train['Physical-Weight'].notna() & train['Physical-Height'].notna(),\n    train['Physical-Weight'] / ((train['Physical-Height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)\n\ncalculate_stats(train, wh_cols)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.816384Z","iopub.execute_input":"2024-12-22T04:38:19.816674Z","iopub.status.idle":"2024-12-22T04:38:19.852328Z","shell.execute_reply.started":"2024-12-22T04:38:19.816647Z","shell.execute_reply":"2024-12-22T04:38:19.851521Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nhiều giá trị có vẻ nằm ngoài phạm vi bình thường... đặc biệt là giá trị tối đa của trọng lượng (142kg) và vòng eo (127cm).","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# Physical-Weight by Age\nplt.subplot(1, 3, 1)\nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Weight', data=train)\nplt.title('Physical-Weight by Age')\nplt.xlabel('Age')\nplt.ylabel('Weight (kg)')\n\n# Physical-Height by Age\nplt.subplot(1, 3, 2)\nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Height', data=train)\nplt.title('Physical-Height by Age')\nplt.xlabel('Age')\nplt.ylabel('Height (cm)')\n\n# Physical-Waist_Circumference vs Physical-Weight\nplt.subplot(1, 3, 3)\nsns.scatterplot(x='Physical-Weight', y='Physical-Waist_Circumference', data=train)\nplt.title('Waist Circumference vs Weight')\nplt.xlabel('Weight (kg)')\nplt.ylabel('Waist Circumference (cm)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:19.853218Z","iopub.execute_input":"2024-12-22T04:38:19.853522Z","iopub.status.idle":"2024-12-22T04:38:21.072185Z","shell.execute_reply.started":"2024-12-22T04:38:19.853492Z","shell.execute_reply":"2024-12-22T04:38:21.071150Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ví dụ lọc các giá trị bất hợp lý\noutliers_logic = train[((train['Physical-Weight'] <= 42) & (train['Physical-Waist_Circumference'] >= 96)) |\n                    ((train['Physical-Height'] >= 170) & (train['Basic_Demos-Age'] <= 7))]\n\ndisplay(outliers_logic)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.073392Z","iopub.execute_input":"2024-12-22T04:38:21.073758Z","iopub.status.idle":"2024-12-22T04:38:21.098734Z","shell.execute_reply.started":"2024-12-22T04:38:21.073724Z","shell.execute_reply":"2024-12-22T04:38:21.097941Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**💡 Insight:** <ul style=\"list-style:circle\"> <li>Trọng lượng và chiều cao đều tăng theo độ tuổi, và vòng eo và trọng lượng có mối tương quan cao, như mong đợi.</li> <li>Tuy nhiên, có những cá nhân cao bất thường so với nhóm tuổi của họ hoặc bị thừa cân nghiêm trọng.</li> <li>Cũng có một vài điểm ngoại lai trong các phép đo vòng eo, có thể là những sai sót (ví dụ: vòng eo 100 cm với trọng lượng 40 kg).</li> <li>Vấn đề với việc làm sạch dữ liệu ở đây là chúng ta không thể đoán được dữ liệu nào là chính xác. Ví dụ, chúng ta có thể thấy một sự kết hợp không thực tế giữa vòng eo 100 cm và trọng lượng 40 kg của một người tham gia, nhưng sai sót nằm ở vòng eo hay trọng lượng? Hoặc một chiều cao khoảng 175 cm của một đứa trẻ 7 tuổi... liệu chiều cao hay độ tuổi đã được nhập sai? Hay đây là dữ liệu chính xác và đứa trẻ mắc chứng vĩ nhân hoặc một rối loạn liên quan đến hormone tăng trưởng?</li> </ul> </div>","metadata":{}},{"cell_type":"markdown","source":"## Blood Pressure & Heart Rate","metadata":{}},{"cell_type":"markdown","source":"\nCó 1000% dữ liệu sai trong các cột Huyết áp/Tần số tim vì các giá trị tối thiểu là nguy hiểm đến tính mạng con người. Chúng ta có thể làm sạch những sai sót kiểu này.","metadata":{}},{"cell_type":"code","source":"bp_hr_cols = [\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP',\n    'Physical-HeartRate'\n]","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.099555Z","iopub.execute_input":"2024-12-22T04:38:21.099766Z","iopub.status.idle":"2024-12-22T04:38:21.116328Z","shell.execute_reply.started":"2024-12-22T04:38:21.099739Z","shell.execute_reply":"2024-12-22T04:38:21.115718Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train[bp_hr_cols] < 50).sum()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.116913Z","iopub.execute_input":"2024-12-22T04:38:21.117196Z","iopub.status.idle":"2024-12-22T04:38:21.136371Z","shell.execute_reply.started":"2024-12-22T04:38:21.117163Z","shell.execute_reply":"2024-12-22T04:38:21.135659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train[(train['Physical-HeartRate'] < 50)])","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.137114Z","iopub.execute_input":"2024-12-22T04:38:21.137366Z","iopub.status.idle":"2024-12-22T04:38:21.174527Z","shell.execute_reply.started":"2024-12-22T04:38:21.137332Z","shell.execute_reply":"2024-12-22T04:38:21.173700Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\nĐây chắc chắn là những phép đo sai. Nhưng một lần nữa, chúng ta không thể chắc chắn thông tin nào là chính xác, vì vậy chúng ta có thể đánh dấu các dòng này để kiểm tra thủ công từng cái một, hoặc thay thế tất cả các giá trị nghi ngờ bằng NaN. Đối với phân tích này, tôi chỉ loại bỏ các giá trị 0 và cả huyết áp nếu huyết áp tâm thu thấp hơn hoặc bằng huyết áp tâm trương.","metadata":{}},{"cell_type":"markdown","source":"## Internet Usage and `sii` (Mức độ sử dụng internet và kết quả dự đoán `sii`)","metadata":{}},{"cell_type":"markdown","source":"Mô tả cuộc thi cho biết mục tiêu là: phát hiện các chỉ số sớm của việc sử dụng Internet và công nghệ có vấn đề (PIU), trong khi định nghĩa của PIU bao gồm việc sử dụng Internet quá mức:\n> PIU là một thuật ngữ bao quát bao gồm một loạt các hành vi trực tuyến có thể gây hại, mang tính lặp lại và không kiểm soát, đến mức chúng được ưu tiên hơn các sở thích khác trong cuộc sống và tiếp tục tồn tại mặc dù có hậu quả tiêu cực.\n\n*[Fendel, J. C., Vogt, A., Brandtner, A., & Schmidt, S. (2024). Mindfulness programs for problematic usage of the internet: A systematic review and meta-analysis. Journal of behavioral addictions, 13(2), 327–353.](https://doi.org/10.1556/2006.2024.00024)*\n\nVậy hãy xem những người tham gia với các điểm số suy giảm khác nhau (SII) đã dành bao nhiêu thời gian trực tuyến trong bộ dữ liệu này.","metadata":{}},{"cell_type":"code","source":"sii_reported = train[train['sii'] != \"Missing\"]\n# sii_reported.loc[:, 'sii'] = sii_reported['sii'].cat.remove_unused_categories()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:21.175298Z","iopub.execute_input":"2024-12-22T04:38:21.175578Z","iopub.status.idle":"2024-12-22T04:38:21.182037Z","shell.execute_reply.started":"2024-12-22T04:38:21.175534Z","shell.execute_reply":"2024-12-22T04:38:21.181203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_reported.groupby(\n    ['internet_use_encoded', 'sii']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:21.182927Z","iopub.execute_input":"2024-12-22T04:38:21.183228Z","iopub.status.idle":"2024-12-22T04:38:21.207717Z","shell.execute_reply.started":"2024-12-22T04:38:21.183193Z","shell.execute_reply":"2024-12-22T04:38:21.206778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_reported.groupby(\n    ['sii', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, sii_group in enumerate(stats.index):\n    group_counts = stats.loc[sii_group] / stats.loc[sii_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"), labeldistance=1.1\n    )\n    axes[i].set_title(f'Hours of using computer/internet\\n for SII = {sii_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:21.208624Z","iopub.execute_input":"2024-12-22T04:38:21.208887Z","iopub.status.idle":"2024-12-22T04:38:21.782595Z","shell.execute_reply.started":"2024-12-22T04:38:21.208856Z","shell.execute_reply":"2024-12-22T04:38:21.781679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_reported.groupby(\n    ['sii', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:21.783540Z","iopub.execute_input":"2024-12-22T04:38:21.783765Z","iopub.status.idle":"2024-12-22T04:38:21.801881Z","shell.execute_reply.started":"2024-12-22T04:38:21.783745Z","shell.execute_reply":"2024-12-22T04:38:21.800973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\n    (train['internet_use_encoded'] == '< 1h/day') & \n    (train['sii'].isin(['2 (Moderate)', '3 (Severe)']))\n]['Basic_Demos-Age'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:21.802738Z","iopub.execute_input":"2024-12-22T04:38:21.803016Z","iopub.status.idle":"2024-12-22T04:38:21.823397Z","shell.execute_reply.started":"2024-12-22T04:38:21.802966Z","shell.execute_reply":"2024-12-22T04:38:21.822615Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**💡 Insight**: <ul style=\"list-style:circle\"> <li>Trong các biểu đồ hộp, mặc dù có sự chồng lấn đáng kể giữa các nhóm SII khác nhau và các danh mục sử dụng Internet, chúng ta thấy một xu hướng tích cực giữa suy giảm PIU và việc sử dụng Internet, với những người có điểm SII cao hơn dành nhiều thời gian trực tuyến hơn (sẽ là điều kỳ lạ nếu điều này không xảy ra, vì việc sử dụng Internet quá mức là điều được giả định theo định nghĩa của PIU).</li> <li>Tuy nhiên, khi mối quan hệ giữa PCIAT_Total và giờ sử dụng Internet được phân tách thêm theo nhóm tuổi (biểu đồ hộp dưới cùng), mối quan hệ phi tuyến giữa tuổi tác, sử dụng Internet và PIU nổi lên, với lứa tuổi thanh thiếu niên là nhóm bị ảnh hưởng nhiều nhất trong tất cả các danh mục sử dụng Internet.</li> <li>Các biểu đồ hình tròn cũng cho thấy có một tỷ lệ đáng kể người tham gia (tổng cộng 83 người), ở mọi độ tuổi, dành rất ít thời gian trực tuyến (dưới 1 giờ mỗi ngày) nhưng có điểm SII cao (20,7% với SII 2 - suy giảm vừa phải và 14,7% với SII = 3 - suy giảm nghiêm trọng).</li> </ul> </div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border: 2px solid #c9c9c9; padding: 15px; border-radius: 5px; background-color: #f7f7f7;\"> <h3>Tóm tắt kết quả</h3> <ol> <li>Điểm SII có xu hướng tăng theo độ tuổi nhưng thể hiện mối quan hệ hình chữ U, với thanh thiếu niên có điểm PCIAT trung vị cao nhất.</li> <li>Càng lớn tuổi, người tham gia càng dành nhiều giờ trực tuyến (xu hướng tuyến tính rõ ràng).</li> <li>Những người có điểm SII cao hơn nói chung dành nhiều thời gian trực tuyến hơn, nhưng thanh thiếu niên nổi bật là nhóm tuổi bị ảnh hưởng nhiều nhất trong tất cả các danh mục sử dụng Internet.</li> <li>Có những người tham gia ở gần như mọi độ tuổi (từ 5 đến 21) dành ít hơn một giờ mỗi ngày trực tuyến và có điểm SII cao.</li> </ol> <p><em>Ghi chú:</em> Các kết quả này cần được diễn giải cẩn thận, vì có sự chồng lấn đáng kể giữa các danh mục SII và sử dụng Internet khác nhau, và các trường hợp nghiêm trọng cũng như người lớn bị thiếu đại diện trong dữ liệu.</p> </div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border: 2px solid #c9c9c9; padding: 15px; border-radius: 5px; background-color: #f7f7f7;\"> <h3>Diễn giải</h3> <p>Mối quan hệ giữa SII và thời gian sử dụng Internet không đơn giản như người ta có thể mong đợi, xét theo định nghĩa của PIU (hay còn gọi là nghiện Internet). Điều này là vì các yếu tố ngoài số giờ sử dụng Internet còn ảnh hưởng đến SII, và phân tích trên cho thấy độ tuổi là yếu tố đặc biệt quan trọng: thanh thiếu niên có vẻ có SII cao nhất trong tất cả các mức độ sử dụng Internet... Nhưng làm thế nào để giải thích điều này: họ có dễ bị PIU hơn không, hay liệu bảng câu hỏi này chỉ nhạy cảm hơn với PIU ở nhóm tuổi này? Hãy thử hiểu rõ hơn về những gì mà biến mục tiêu của chúng ta phản ánh.</p> <p>Các câu hỏi trong bảng câu hỏi PCIAT (dùng để xác định SII, xem data_dict.csv) có vẻ được thiết kế để đo lường các tác động cảm xúc và xã hội liên quan đến việc sử dụng Internet (phụ thuộc cảm xúc vào Internet, cô lập xã hội, bỏ bê trách nhiệm, và tác động của việc sử dụng Internet lên các mối quan hệ và tâm trạng). Nói cách khác, mục đích là đo lường mức độ vấn đề của hành vi liên quan đến việc sử dụng Internet. Tuy nhiên, nhận thức của cha mẹ tự nhiên bị thiên lệch và bị ảnh hưởng bởi nhiều yếu tố - như thói quen sử dụng Internet của chính họ, thái độ văn hóa, hoặc những mong muốn/ý tưởng của họ về cách con cái họ nên cư xử.</p> <p>Hơn nữa, bạn có thể tưởng tượng rằng việc dành ít hơn một giờ mỗi ngày trên Internet có thể dẫn đến các vấn đề như căng thẳng cảm xúc, bỏ bê trách nhiệm hay rút lui khỏi gia đình không? Tôi không nghĩ rằng một giờ mỗi ngày trên Internet với bất kỳ nội dung nào có thể dẫn đến những điều đó... Sự hiện diện của điều này trong dữ liệu chỉ chứng minh rằng người trả lời không thành thật khi trả lời các câu hỏi trong PCIAT và việc sử dụng Internet, hoặc điểm SII bị ảnh hưởng bởi các yếu tố khác không liên quan đến PIU.</p> <p>Điều này chỉ ra rằng yếu tố duy nhất liên kết tất cả các dữ liệu khác mà chúng ta có (hoạt động thể chất, dữ liệu gia tốc, giấc ngủ, v.v.) với việc sử dụng Internet (và chúng ta cần sự kết nối này để dự đoán tác động của PIU) có thể không đáng tin cậy và thiên lệch, giống như biến mục tiêu.</p> <p>Điều này ám chỉ rằng người tham gia có thể có những hành vi xã hội hoặc tâm trạng trước đó không liên quan đến việc sử dụng Internet (và PIU) tự thân. Bảng câu hỏi là một công cụ chủ quan, ngay cả khi được hoàn thành bởi cha mẹ, trong khi tuổi thanh thiếu niên là một giai đoạn nổi bật trong cuộc sống - thời kỳ hình thành bản sắc, mối quan hệ bạn bè đang thay đổi và tìm kiếm sự độc lập - tất cả những điều này có thể làm tăng cường các hành vi như thay đổi tâm trạng, không vâng lời và tính bốc đồng. Do đó, SII có thể đang ghi lại mức độ vấn đề của những hành vi phát triển rộng hơn này thay vì chỉ riêng việc sử dụng Internet.</p> <p>Thêm vào đó, tính khả thi của bảng câu hỏi trên phạm vi độ tuổi là điều cần phải xem xét. Tôi nghĩ rằng tất cả các câu hỏi trong PCIAT phù hợp hơn với thanh thiếu niên. Ví dụ:</p> <ul style=\"list-style:circle\"> <li>Trẻ em 5-7 tuổi có thể không có công việc nhà, vì điều này phụ thuộc vào các chuẩn mực văn hóa.</li> <li>Câu hỏi về tác động học tập có thể không áp dụng với trẻ em chưa đi học hoặc người lớn đã tốt nghiệp.</li> <li>Việc sử dụng email và nhận cuộc gọi điện thoại từ \"bạn bè trực tuyến\" có vẻ không phù hợp với trẻ nhỏ.</li> <li>Các câu hỏi về phản ứng với thời gian cho phép dành cho việc sử dụng Internet (ít nhất có 3 câu hỏi) thường không áp dụng cho người lớn.</li> </ul> <p>Các câu hỏi không áp dụng cho độ tuổi của người tham gia có thể dẫn đến những câu trả lời lệch lạc hoặc không liên quan. Tất cả những điều này thách thức tính hợp lệ cấu trúc của SII và đặt câu hỏi liệu nó có đo lường chính xác PIU hay bị ảnh hưởng bởi các yếu tố hành vi khác.</p> </div>","metadata":{}},{"cell_type":"markdown","source":"## FitnessGram Vitals and Treadmill","metadata":{}},{"cell_type":"code","source":"groups.get('FitnessGram Vitals and Treadmill', [])","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.824218Z","iopub.execute_input":"2024-12-22T04:38:21.824426Z","iopub.status.idle":"2024-12-22T04:38:21.839982Z","shell.execute_reply.started":"2024-12-22T04:38:21.824407Z","shell.execute_reply":"2024-12-22T04:38:21.839231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[train['Fitness_Endurance-Max_Stage'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with Fitness_Endurance-Max_Stage data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.840829Z","iopub.execute_input":"2024-12-22T04:38:21.841119Z","iopub.status.idle":"2024-12-22T04:38:21.858803Z","shell.execute_reply.started":"2024-12-22T04:38:21.841091Z","shell.execute_reply":"2024-12-22T04:38:21.857773Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nsns.violinplot(x='Basic_Demos-Age', y='Fitness_Endurance-Max_Stage', data=train, palette=\"Set3\")\nplt.title('Fitness Endurance Max Stage by Age')\nplt.xlabel('Age')\nplt.ylabel('Max Stage')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:21.859824Z","iopub.execute_input":"2024-12-22T04:38:21.860119Z","iopub.status.idle":"2024-12-22T04:38:22.466717Z","shell.execute_reply.started":"2024-12-22T04:38:21.860096Z","shell.execute_reply":"2024-12-22T04:38:22.465880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = [\n    'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec'\n]\ncalculate_stats(train, cols)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.467590Z","iopub.execute_input":"2024-12-22T04:38:22.467887Z","iopub.status.idle":"2024-12-22T04:38:22.488143Z","shell.execute_reply.started":"2024-12-22T04:38:22.467857Z","shell.execute_reply":"2024-12-22T04:38:22.487298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Check the combinations of missing values","metadata":{}},{"cell_type":"code","source":"train[\n    (train['Fitness_Endurance-Max_Stage'].notna()) & \n    (train['Fitness_Endurance-Time_Mins'].isna() | \n     train['Fitness_Endurance-Time_Sec'].isna())\n][cols]","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.489109Z","iopub.execute_input":"2024-12-22T04:38:22.489397Z","iopub.status.idle":"2024-12-22T04:38:22.500143Z","shell.execute_reply.started":"2024-12-22T04:38:22.489368Z","shell.execute_reply":"2024-12-22T04:38:22.499280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Có thể trong quá trình nhập dữ liệu, phút hoặc giây bị bỏ trống (được nhập là NaN) khi lẽ ra chúng phải được ghi là 0 phút/giây. Mặc dù giây bị thiếu không quan trọng lắm, nhưng phút bị thiếu có thể thực sự là bị thiếu và việc xử lý chúng như 0 sẽ cho kết quả kiểm tra không chính xác. Tôi nghĩ tốt hơn là nên loại bỏ những trường hợp nghi ngờ này.","metadata":{}},{"cell_type":"code","source":"train.loc[\n    (train['Fitness_Endurance-Max_Stage'].notna()) & \n    (train['Fitness_Endurance-Time_Mins'].isna() | \n     train['Fitness_Endurance-Time_Sec'].isna()), cols\n] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.500813Z","iopub.execute_input":"2024-12-22T04:38:22.500991Z","iopub.status.idle":"2024-12-22T04:38:22.516781Z","shell.execute_reply.started":"2024-12-22T04:38:22.500975Z","shell.execute_reply":"2024-12-22T04:38:22.516207Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Get one time column (mins + sec)","metadata":{}},{"cell_type":"code","source":"train['Fitness_Endurance-Total_Time_Sec'] = train[\n    'Fitness_Endurance-Time_Mins'\n] * 60 + train['Fitness_Endurance-Time_Sec']","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.517603Z","iopub.execute_input":"2024-12-22T04:38:22.517887Z","iopub.status.idle":"2024-12-22T04:38:22.534124Z","shell.execute_reply.started":"2024-12-22T04:38:22.517866Z","shell.execute_reply":"2024-12-22T04:38:22.533508Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Recalculate stats:","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, ['Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Total_Time_Sec'])","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.534745Z","iopub.execute_input":"2024-12-22T04:38:22.534924Z","iopub.status.idle":"2024-12-22T04:38:22.568764Z","shell.execute_reply.started":"2024-12-22T04:38:22.534908Z","shell.execute_reply":"2024-12-22T04:38:22.568152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**💡 Insight:** <ul style=\"list-style:circle\"> <li>Trung bình, các người tham gia đạt đến giai đoạn 5 trong bài kiểm tra sức bền.</li> <li>Một số người tham gia không hoàn thành giai đoạn đầu tiên (min = 0), hoặc đây lại là lỗi trong dữ liệu.</li> <li>Có một số ít người tham gia có sức bền đặc biệt cao ở độ tuổi 7-8.</li> <li>Có một lượng dữ liệu thiếu đáng kể (hơn 80% bộ dữ liệu thiếu thông tin này).</li> </ul> </div>","metadata":{}},{"cell_type":"markdown","source":"## Physical Measures","metadata":{}},{"cell_type":"code","source":"features_physical = groups.get('Physical Measures', [])\ncols = [col for col in features_physical if col in continuous_cols]\n\nplt.figure(figsize=(24, 10))\nn_cols = 4\nn_rows = len(cols) // n_cols + 1\n\nfor i, col in enumerate(cols):\n    plt.subplot(n_rows, n_cols, i + 1)\n    train[col].hist(bins=20)\n    plt.title(col)\n\nplt.subplot(n_rows, n_cols, len(cols) + 1)\nseason_counts = train['Physical-Season'].value_counts(dropna=False)\nplt.pie(\n    season_counts,\n    labels=season_counts.index,\n    autopct='%1.1f%%',\n    startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('Physical-Season')\n\nplt.suptitle('Histograms for Physical Measures and Physical-Season Pie Chart', y=1.05)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:22.569475Z","iopub.execute_input":"2024-12-22T04:38:22.569682Z","iopub.status.idle":"2024-12-22T04:38:24.735436Z","shell.execute_reply.started":"2024-12-22T04:38:22.569663Z","shell.execute_reply":"2024-12-22T04:38:24.734511Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[cols] = train[cols].replace(0, np.nan)\ntrain.loc[train['Physical-Systolic_BP'] <= train['Physical-Diastolic_BP'], bp_hr_cols] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:24.736532Z","iopub.execute_input":"2024-12-22T04:38:24.736837Z","iopub.status.idle":"2024-12-22T04:38:24.746871Z","shell.execute_reply.started":"2024-12-22T04:38:24.736807Z","shell.execute_reply":"2024-12-22T04:38:24.745945Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, cols)","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:24.747670Z","iopub.execute_input":"2024-12-22T04:38:24.747922Z","iopub.status.idle":"2024-12-22T04:38:24.788514Z","shell.execute_reply.started":"2024-12-22T04:38:24.747903Z","shell.execute_reply":"2024-12-22T04:38:24.787746Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bio-electric Impedance Analysis","metadata":{}},{"cell_type":"code","source":"data_dict[data_dict['Instrument'] == 'Bio-electric Impedance Analysis']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:24.789235Z","iopub.execute_input":"2024-12-22T04:38:24.789538Z","iopub.status.idle":"2024-12-22T04:38:24.800432Z","shell.execute_reply.started":"2024-12-22T04:38:24.789506Z","shell.execute_reply":"2024-12-22T04:38:24.799542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There is no information in the competition description about what equipment was used, is this raw data or did they use some BIA equation models to estimate the parameters. But it's likely that the BIA data has already been processed using a BIA equation model. It is very important to note that BIA is not a precise method, for example it tends to overestimate muscle mass, so equations have been developed to estimate muscle mass based on factors such as age, sex, height, weight and resistance and/or reactance estimated by BIA... a large number of prediction equation models have been generated through various validation studies ([link](https://clinicalnutritionespen.com/article/S2405-4577(19)30478-4/fulltext)). It is essential that all recordings are processed with the same equation, but we cannot be sure. ","metadata":{}},{"cell_type":"code","source":"bia_data_dict = data_dict[data_dict['Instrument'] == 'Bio-electric Impedance Analysis']\ncategorical_columns = bia_data_dict[bia_data_dict['Type'] == 'categorical int']['Field'].tolist()\ncontinuous_columns = bia_data_dict[bia_data_dict['Type'] == 'float']['Field'].tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:24.807722Z","iopub.execute_input":"2024-12-22T04:38:24.808024Z","iopub.status.idle":"2024-12-22T04:38:24.820130Z","shell.execute_reply.started":"2024-12-22T04:38:24.807983Z","shell.execute_reply":"2024-12-22T04:38:24.819361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(24, 20))\n\nfor idx, col in enumerate(continuous_columns):\n    plt.subplot(4, 4, idx + 1)\n    sns.histplot(train[col].dropna(), bins=20, kde=True)\n    plt.title(data_dict[data_dict['Field'] == col]['Description'].values[0])\n    plt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:24.821261Z","iopub.execute_input":"2024-12-22T04:38:24.821525Z","iopub.status.idle":"2024-12-22T04:38:29.629297Z","shell.execute_reply.started":"2024-12-22T04:38:24.821502Z","shell.execute_reply":"2024-12-22T04:38:29.628403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, continuous_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:29.630155Z","iopub.execute_input":"2024-12-22T04:38:29.630446Z","iopub.status.idle":"2024-12-22T04:38:29.677911Z","shell.execute_reply.started":"2024-12-22T04:38:29.630414Z","shell.execute_reply":"2024-12-22T04:38:29.677056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**💡 Insight:** <ul style=\"list-style:circle\"> <li>Phân bố của các phép đo phân tích trở kháng sinh học (bioelectrical impedance analysis) trong tập dữ liệu cho thấy hầu hết chúng không hữu ích: **phân bố lệch nhiều**, với phần lớn người tham gia có giá trị biên và một số ít giá trị ngoại lai (có thể là lỗi đo lường).</li> <li>Một số biến, chẳng hạn như Chỉ số Khối Lượng Mỡ (Fat Mass Index) và Tỷ Lệ Mỡ Cơ Thể (Body Fat Percentage), xuất hiện các giá trị **âm** không hợp lý, và gần như tất cả các giá trị cao cực đoan, cho thấy có thể có vấn đề về chất lượng dữ liệu.</li> <li>Điều đáng chú ý là trong khi kỷ lục về body fat percentage được ghi nhận là **59.9%** thì lại có người trên 60%, thậm chí là 150%. Điều này là hết sức vô lý </li></ul> </div>","metadata":{}},{"cell_type":"markdown","source":"# 5. Xử lý và chuẩn bị dữ liệu:","metadata":{}},{"cell_type":"markdown","source":"Các giá trị không hợp lý, chẳng hạn như fat body percentage trên 60% hoặc negative bone mineral content bị âm, đã được loại bỏ và thay thế bằng NaN.","metadata":{}},{"cell_type":"code","source":"def clean_features(df):\n    # Remove highly implausible values\n\n    # Clip Grip\n    df[['FGC-FGC_GSND', 'FGC-FGC_GSD']] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].clip(lower=9, upper=60)\n    # Remove implausible body-fat\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] < 5, np.nan, df[\"BIA-BIA_Fat\"])\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] > 60, np.nan, df[\"BIA-BIA_Fat\"])\n    # Basal Metabolic Rate\n    df[\"BIA-BIA_BMR\"] = np.where(df[\"BIA-BIA_BMR\"] > 4000, np.nan, df[\"BIA-BIA_BMR\"])\n    # Daily Energy Expenditure\n    df[\"BIA-BIA_DEE\"] = np.where(df[\"BIA-BIA_DEE\"] > 8000, np.nan, df[\"BIA-BIA_DEE\"])\n    # Bone Mineral Content\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] <= 0, np.nan, df[\"BIA-BIA_BMC\"])\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] > 10, np.nan, df[\"BIA-BIA_BMC\"])\n    # Fat Free Mass Index\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] <= 0, np.nan, df[\"BIA-BIA_FFM\"])\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] > 300, np.nan, df[\"BIA-BIA_FFM\"])\n    # Fat Mass Index\n    df[\"BIA-BIA_FMI\"] = np.where(df[\"BIA-BIA_FMI\"] < 0, np.nan, df[\"BIA-BIA_FMI\"])\n    # Extra Cellular Water\n    df[\"BIA-BIA_ECW\"] = np.where(df[\"BIA-BIA_ECW\"] > 100, np.nan, df[\"BIA-BIA_ECW\"])\n    # Intra Cellular Water\n    # df[\"BIA-BIA_ICW\"] = np.where(df[\"BIA-BIA_ICW\"] > 100, np.nan, df[\"BIA-BIA_ICW\"])\n    # Lean Dry Mass\n    df[\"BIA-BIA_LDM\"] = np.where(df[\"BIA-BIA_LDM\"] > 100, np.nan, df[\"BIA-BIA_LDM\"])\n    # Lean Soft Tissue\n    df[\"BIA-BIA_LST\"] = np.where(df[\"BIA-BIA_LST\"] > 300, np.nan, df[\"BIA-BIA_LST\"])\n    # Skeletal Muscle Mass\n    df[\"BIA-BIA_SMM\"] = np.where(df[\"BIA-BIA_SMM\"] > 300, np.nan, df[\"BIA-BIA_SMM\"])\n    # Total Body Water\n    df[\"BIA-BIA_TBW\"] = np.where(df[\"BIA-BIA_TBW\"] > 300, np.nan, df[\"BIA-BIA_TBW\"])\n    \n    return df\n\ntrain = clean_features(train)\ntest = clean_features(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:38:29.678824Z","iopub.execute_input":"2024-12-22T04:38:29.679117Z","iopub.status.idle":"2024-12-22T04:38:29.699938Z","shell.execute_reply.started":"2024-12-22T04:38:29.679081Z","shell.execute_reply":"2024-12-22T04:38:29.699203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Trực quan hóa dữ liệu:","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    # Loại bỏ các cột về mùa\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n\n    # Tạo các feature mới\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    # tỷ lệ phần trăm mỡ cơ thể (BFP) chia cho chỉ số BMI\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    # khối lượng nạc không mỡ (Fat-Free Mass Index - FFMI) và phần trăm mỡ cơ thể\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    # chỉ số khối mỡ cơ thể (FMI) và phần trăm mỡ cơ thể\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    # Tính tỷ lệ giữa khối lượng mô mềm nạc (Lean Soft Tissue - LST) và tổng lượng nước cơ thể (Total Body Water - TBW)\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    # tổng năng lượng chuyển hóa cơ bản (BMR) nhân với phần trăm mỡ cơ thể (BFP)\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    # tổng lượng năng lượng tiêu hao hàng ngày (DEE) nhân với phần trăm mỡ cơ thể (BFP)\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    # tỷ lệ giữa năng lượng chuyển hóa cơ bản (BMR) và cân nặng\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    # tỷ lệ giữa tiêu hao năng lượng hàng ngày (DEE) và cân nặng\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    # Tính tỷ lệ giữa khối lượng cơ xương (Skeletal Muscle Mass - SMM) và chiều cao\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    # tỷ lệ giữa khối lượng cơ xương (SMM) và chỉ số khối mỡ cơ thể (FMI)\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    # tỷ lệ giữa tổng lượng nước trong cơ thể (TBW) và cân nặng\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    # tỷ lệ giữa nước nội bào (Intracellular Water - ICW) và tổng lượng nước cơ thể (TBW)\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    # BMI với nhịp tim\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n\n    # Để tránh phép chia cho 0, thêm 1 vào giờ sử dụng internet\n    hoursDayAdd1hour = df['PreInt_EduHx-computerinternet_hoursday'] + 1\n\n    # Tương tác giữa thời gian sử dụng internet và mức độ mất ngủ\n    df['Internet_Hours_SDS'] = hoursDayAdd1hour * df['SDS-SDS_Total_T']\n    \n    # Tương tác giữa thời gian sử dụng internet và tổng điểm hoạt động\n    df['Internet_Hours_PAQ'] = hoursDayAdd1hour * df['PAQ_C-PAQ_C_Total']\n\n    # Tỉ lệ thời gian sử dụng internet so với hoạt động\n    df['Internet_to_Activity_Ratio'] = hoursDayAdd1hour / df['PAQ_C-PAQ_C_Total']\n\n    # Tỉ lệ nhịp tim và tuổi\n    df['HeartRate_Age_Ratio'] = df['Physical-HeartRate'] / df['Basic_Demos-Age']\n\n    # Tương tác nhịp tim và mức độ mất ngủ\n    df['HeartRate_SDS'] = df['Physical-HeartRate'] * df['SDS-SDS_Total_T']\n\n    # Tỉ lệ mất ngủ và điểm đánh giá chức năng chung\n    df['SDS_CGAS_Ratio'] = df['SDS-SDS_Total_T'] / df['CGAS-CGAS_Score']\n\n    # mất ngủ * hoạt động\n    df['SDS_PAQ_Interaction'] = df['SDS-SDS_Total_T'] * df['PAQ_C-PAQ_C_Total']\n\n    # Tỉ lệ thời gian sử dụng internet và nhịp tim\n    df['Internet_Hours_HeartRate_Ratio'] = hoursDayAdd1hour / df['Physical-HeartRate']\n\n    # Tương tác điểm chức năng chung và hoạt động\n    df['CGAS_PAQ_Interaction'] = df['CGAS-CGAS_Score'] * df['PAQ_C-PAQ_C_Total']\n\n    # mất ngủ và nhịp tim\n    df['SDS_HeartRate_Ratio'] = df['SDS-SDS_Total_T'] / df['Physical-HeartRate']\n    # mất ngủ cân nặng\n    df['SDS_Weight_Interaction'] = df['SDS-SDS_Total_T'] * df['Physical-Weight']\n    # mất ngủ và eo\n    df['SDS_Waist_Interaction'] = df['SDS-SDS_Total_T'] * df['Physical-Waist_Circumference']\n    # tổng thời gian sec\n    df['Fitness_Endurance-Total_Time_Sec'] = df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec']\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:29.700682Z","iopub.execute_input":"2024-12-22T04:38:29.700870Z","iopub.status.idle":"2024-12-22T04:38:29.716807Z","shell.execute_reply.started":"2024-12-22T04:38:29.700853Z","shell.execute_reply":"2024-12-22T04:38:29.716058Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef plot_correlation_heatmap(train):\n    # Danh sách các cột liên quan đến SDS-SDS_Total_T\n    columns_to_correlate = [\n        'SDS-SDS_Total_T', 'Internet_Hours_SDS', 'HeartRate_SDS', 'SDS_CGAS_Ratio', \n        'SDS_PAQ_Interaction', 'SDS_HeartRate_Ratio', \n        'SDS_Weight_Interaction', 'SDS_Waist_Interaction'\n    ]\n    \n    # Lọc các cột có trong DataFrame\n    available_columns = [col for col in columns_to_correlate if col in train.columns]\n    \n    # Tính toán ma trận tương quan\n    corr_matrix = train[available_columns].corr()\n    \n    # Vẽ heatmap\n    plt.figure(figsize=(10, 8))\n    sns.heatmap(corr_matrix, annot=True, cmap='viridis', fmt='.2f', cbar=True)\n    plt.title('Heatmap of Correlations with SDS-SDS_Total_T')\n    plt.tight_layout()\n    plt.show()\n\n# Giả sử `train` là DataFrame sau khi đã qua hàm feature_engineering\ntrain = feature_engineering(train)\nplot_correlation_heatmap(train)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:29.717660Z","iopub.execute_input":"2024-12-22T04:38:29.717937Z","iopub.status.idle":"2024-12-22T04:38:30.227934Z","shell.execute_reply.started":"2024-12-22T04:38:29.717917Z","shell.execute_reply":"2024-12-22T04:38:30.227074Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Lọc các cột số học trong train\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\n\n# Loại trừ các cột bắt đầu bằng 'PCIAT' và cột có đuôi 'zone'\nnumeric_cols_filtered = [col for col in numeric_cols if not col.startswith('PCIAT') and not col.endswith('Zone')]\n\n# Thêm cột 'PCIAT-PCIAT_Total' nếu nó không có trong danh sách đã lọc\nif 'PCIAT-PCIAT_Total' not in numeric_cols_filtered:\n    numeric_cols_filtered.append('PCIAT-PCIAT_Total')\n\n# Lọc dữ liệu chỉ còn các cột đã chọn\ntrain_filtered = train[numeric_cols_filtered]\n\n# Tính toán ma trận tương quan\ncorr_matrix = train_filtered.corr()\n\n# Cột 'SDS-SDS_Total_Raw' phải có mặt trong mỗi heatmap\ntarget_col = 'SDS-SDS_Total_Raw'\n\n# Kiểm tra nếu cột này có trong ma trận tương quan\nif target_col not in corr_matrix.columns:\n    raise ValueError(f\"Column {target_col} not found in the correlation matrix.\")\n\n# Số cột tối đa trên mỗi heatmap\nmax_cols_per_heatmap = 10\n\n# Sắp xếp các cột ngoại trừ 'SDS-SDS_Total_Raw'\nother_cols = [col for col in corr_matrix.columns if col != target_col]\n\n# Chia ma trận tương quan thành các phần nhỏ hơn, mỗi phần bao gồm 'SDS-SDS_Total_Raw'\nn_cols = len(other_cols)\nn_heatmaps = (n_cols // max_cols_per_heatmap) + 1\n\n# Vẽ nhiều heatmap\nfor i in range(n_heatmaps):\n    start_col = i * max_cols_per_heatmap\n    end_col = min((i + 1) * max_cols_per_heatmap, n_cols)\n    \n    # Lấy phần con của ma trận tương quan bao gồm 'SDS-SDS_Total_Raw'\n    selected_cols = [target_col] + other_cols[start_col:end_col]\n    corr_submatrix = corr_matrix.loc[selected_cols, selected_cols]\n    \n    plt.figure(figsize=(10, 8))\n    sns.heatmap(corr_submatrix, annot=True, cmap='coolwarm', fmt='.2f', linewidths=0.5)\n    plt.title(f'Correlation Matrix (Including {target_col}, Columns {start_col+1} to {end_col})')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:30.228776Z","iopub.execute_input":"2024-12-22T04:38:30.228992Z","iopub.status.idle":"2024-12-22T04:38:35.316613Z","shell.execute_reply.started":"2024-12-22T04:38:30.228972Z","shell.execute_reply":"2024-12-22T04:38:35.315756Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Relationships with the target variable (PCIAT_Total for complete PCIAT responses)","metadata":{}},{"cell_type":"code","source":"data_subset = train[cols + ['PCIAT-PCIAT_Total']]\n\ncorr_matrix = data_subset.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-22T04:38:35.317676Z","iopub.execute_input":"2024-12-22T04:38:35.318023Z","iopub.status.idle":"2024-12-22T04:38:35.858951Z","shell.execute_reply.started":"2024-12-22T04:38:35.317970Z","shell.execute_reply":"2024-12-22T04:38:35.858165Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Nhận xét và ghi chú:","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}