{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Based\nhttps://www.kaggle.com/code/ambrosm/piu-eda-which-makes-sense ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.324384Z","iopub.execute_input":"2024-12-08T17:19:05.325213Z","iopub.status.idle":"2024-12-08T17:19:05.333959Z","shell.execute_reply.started":"2024-12-08T17:19:05.325148Z","shell.execute_reply":"2024-12-08T17:19:05.331760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.336742Z","iopub.execute_input":"2024-12-08T17:19:05.337468Z","iopub.status.idle":"2024-12-08T17:19:05.357055Z","shell.execute_reply.started":"2024-12-08T17:19:05.337395Z","shell.execute_reply":"2024-12-08T17:19:05.354484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nprint(f\"Shape of train_df: {train_df.shape}\")\nprint(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.361840Z","iopub.execute_input":"2024-12-08T17:19:05.362420Z","iopub.status.idle":"2024-12-08T17:19:05.438113Z","shell.execute_reply.started":"2024-12-08T17:19:05.362363Z","shell.execute_reply":"2024-12-08T17:19:05.436356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.441320Z","iopub.execute_input":"2024-12-08T17:19:05.441737Z","iopub.status.idle":"2024-12-08T17:19:05.451187Z","shell.execute_reply.started":"2024-12-08T17:19:05.441699Z","shell.execute_reply":"2024-12-08T17:19:05.448802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# Loại bỏ hàng có 'sii' rỗng\ntrain_df_filtered = train_df[train_df['sii'].notna()]\ntrain_df_filtered['sii'].count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.452837Z","iopub.execute_input":"2024-12-08T17:19:05.453204Z","iopub.status.idle":"2024-12-08T17:19:05.471069Z","shell.execute_reply.started":"2024-12-08T17:19:05.453171Z","shell.execute_reply":"2024-12-08T17:19:05.469726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Phân phối cột mục tiêu 'sii'\nsii_counts = train_df_filtered['sii'].value_counts()\n\n# Vẽ\nplt.figure(figsize=(8, 6))\nplt.bar(sii_counts.index, sii_counts.values)\nplt.title(\"Phân phối cột mục tiêu 'sii'\")\nplt.xlabel(\"'sii'\")\nplt.ylabel(\"Count\")\nplt.xticks(ticks=sii_counts.index, labels=[int(i) for i in sii_counts.index], fontsize=12)\nplt.grid(axis='y', linestyle='--')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.472774Z","iopub.execute_input":"2024-12-08T17:19:05.473173Z","iopub.status.idle":"2024-12-08T17:19:05.834180Z","shell.execute_reply.started":"2024-12-08T17:19:05.473137Z","shell.execute_reply":"2024-12-08T17:19:05.832664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Số giá trị thiếu theo cột\nmissing_counts = train_df_filtered.isnull().sum()\n\n# Tỷ lệ giá trị thiếu\nmissing_ratio = (train_df_filtered.isnull().sum() / len(train_df_filtered)) * 100\n\n# Sắp xếp các cột theo tỷ lệ giá trị thiếu\nmissing_ratio_sorted = missing_ratio.sort_values(ascending=False)\n\n# Vẽ\nplt.figure(figsize=(10, len(missing_ratio_sorted) * 0.2))\nplt.barh(missing_ratio_sorted.index, missing_ratio_sorted.values)\nplt.xlabel('Phần trăm')\nplt.ylabel('Cột')\nplt.title('Tỷ lệ giá trị thiếu theo cột (%)')\nplt.grid(axis='x', linestyle='--')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:05.835716Z","iopub.execute_input":"2024-12-08T17:19:05.836070Z","iopub.status.idle":"2024-12-08T17:19:07.002285Z","shell.execute_reply.started":"2024-12-08T17:19:05.836040Z","shell.execute_reply":"2024-12-08T17:19:07.000997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loại bỏ những cột có tỷ lệ giá trị thiếu > 50%\ncolumns_to_keep = missing_ratio[missing_ratio <= 50].index\n\ntrain_df_cleaned = train_df_filtered[columns_to_keep]\nprint(f\"Kích thước của train_df_cleaned: {train_df_cleaned.shape}\")\n\ntrain_df_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:07.004113Z","iopub.execute_input":"2024-12-08T17:19:07.004528Z","iopub.status.idle":"2024-12-08T17:19:07.028801Z","shell.execute_reply.started":"2024-12-08T17:19:07.004462Z","shell.execute_reply":"2024-12-08T17:19:07.026280Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Demographics","metadata":{}},{"cell_type":"markdown","source":"#### Enroll Season","metadata":{}},{"cell_type":"code","source":"# Tính số lượng người tham gia của từng mùa\nes = train_df_cleaned['Basic_Demos-Enroll_Season'].value_counts()\n\n# Vẽ\nplt.figure(figsize=(8, 6))\nplt.pie(es.values, labels=es.index, autopct='%1.1f%%', startangle=90)\nplt.title('Season of enrollment')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:07.032677Z","iopub.execute_input":"2024-12-08T17:19:07.033123Z","iopub.status.idle":"2024-12-08T17:19:07.211640Z","shell.execute_reply.started":"2024-12-08T17:19:07.033086Z","shell.execute_reply":"2024-12-08T17:19:07.209845Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Người tham gia phân bổ ở 4 mùa là gần như nhau","metadata":{}},{"cell_type":"markdown","source":"#### Age","metadata":{}},{"cell_type":"code","source":"# Phân bố độ tuổi của nam và nữ tham gia khảo sát\ncolors = ['lightblue', 'coral']\nlabels = ['boys', 'girls']\n\n_, axs = plt.subplots(2, 1, sharex=True)\n\nfor sex, color, label in zip(range(2), colors, labels):\n    es = train_df_cleaned[train_df_cleaned['Basic_Demos-Sex'] == sex]['Basic_Demos-Age'].value_counts().sort_index()\n    axs[sex].bar(es.index, es.values, color=color, label=label)\n    axs[sex].xaxis.set_major_locator(MaxNLocator(integer=True))\n    axs[sex].set_ylabel('count')\n    axs[sex].legend()\n\nfor ax in axs:\n    ax.grid(axis='y', linestyle='--')\naxs[1].set_xlabel('years')\nplt.suptitle('Age distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:07.213599Z","iopub.execute_input":"2024-12-08T17:19:07.214302Z","iopub.status.idle":"2024-12-08T17:19:07.742044Z","shell.execute_reply.started":"2024-12-08T17:19:07.214229Z","shell.execute_reply":"2024-12-08T17:19:07.740680Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Độ tuổi khảo sát: 5 - 22 tuổi","metadata":{}},{"cell_type":"markdown","source":"#### Sex","metadata":{}},{"cell_type":"code","source":"# Tỷ lệ nam, nữ\nsex_mapping = {0: 'boys', 1: 'girls'}\n\nes = train_df_cleaned['Basic_Demos-Sex'].value_counts()\nes.index = es.index.map(sex_mapping)\n\nplt.pie(es, labels=es.index, autopct='%1.1f%%', startangle=90)\nplt.title('Sex of participant')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:07.743459Z","iopub.execute_input":"2024-12-08T17:19:07.743880Z","iopub.status.idle":"2024-12-08T17:19:07.879099Z","shell.execute_reply.started":"2024-12-08T17:19:07.743840Z","shell.execute_reply":"2024-12-08T17:19:07.877726Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Target distribution","metadata":{}},{"cell_type":"code","source":"from matplotlib.ticker import PercentFormatter\n\ncolors = ['lightblue', 'coral']\nlabels = ['boys', 'girls']\nx_labels = ['None', 'Mild', 'Moderate', 'Severe']\n\n# Vẽ\n_, axs = plt.subplots(2, 1, sharex=True, figsize=(8, 6))\n\nfor sex, ax, color, label in zip([0, 1], axs, colors, labels):\n    es = train_df_cleaned[train_df_cleaned['Basic_Demos-Sex'] == sex]['sii'].value_counts(normalize=True).sort_index()\n    ax.bar(es.index, es.values, color=color, label=label)\n    ax.set_ylabel('Percentage')\n    ax.legend()\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    ax.set_xticks(range(len(x_labels)))\n    ax.set_xticklabels(x_labels)\n    ax.grid(axis='y', linestyle='--')\n\naxs[1].set_xlabel(\"'sii'\")\nplt.suptitle('Target distribution')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:07.880901Z","iopub.execute_input":"2024-12-08T17:19:07.881907Z","iopub.status.idle":"2024-12-08T17:19:08.366303Z","shell.execute_reply.started":"2024-12-08T17:19:07.881828Z","shell.execute_reply":"2024-12-08T17:19:08.364698Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tỷ lệ nghiện Internet ở trẻ nam cao hơn so với nữ","metadata":{}},{"cell_type":"markdown","source":"## Fill null-values","metadata":{}},{"cell_type":"markdown","source":"#### Numerical","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# Lấy các cột số\nnum_cols = train_df_cleaned.select_dtypes(include=['float64', 'int64']).columns\n\n# Điền giá trị thiếu\ntrain_df_cleaned.loc[:, num_cols] = knn_imputer.fit_transform(train_df_cleaned[num_cols])\n\ntrain_df_cleaned.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:08.367927Z","iopub.execute_input":"2024-12-08T17:19:08.368326Z","iopub.status.idle":"2024-12-08T17:19:10.296487Z","shell.execute_reply.started":"2024-12-08T17:19:08.368286Z","shell.execute_reply":"2024-12-08T17:19:10.295058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Catalogue","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Lấy danh sách các cột chứa dữ liệu kiểu object\ncat_cols = train_df_cleaned.select_dtypes(include=['object']).columns\n\n# Tạo bộ imputer để điền giá trị thiếu bằng giá trị xuất hiện nhiều nhất\ncat_imputer = SimpleImputer(strategy='most_frequent')\n\n# Sử dụng iloc để cập nhật các cột dữ liệu\ncat_col_indices = [train_df_cleaned.columns.get_loc(col) for col in cat_cols]\ntrain_df_cleaned.iloc[:, cat_col_indices] = cat_imputer.fit_transform(train_df_cleaned.iloc[:, cat_col_indices])\n\ntrain_df_cleaned.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:10.298871Z","iopub.execute_input":"2024-12-08T17:19:10.299293Z","iopub.status.idle":"2024-12-08T17:19:10.330087Z","shell.execute_reply.started":"2024-12-08T17:19:10.299254Z","shell.execute_reply":"2024-12-08T17:19:10.328710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding","metadata":{}},{"cell_type":"code","source":"train_df_cleaned = train_df_cleaned.copy()  # Tạo bản sao độc lập\n\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\nseason_cols = [col for col in train_df_cleaned.columns if 'Season' in col]\n\nfor col in season_cols:\n    train_df_cleaned.loc[:, col] = train_df_cleaned[col].map(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:10.331920Z","iopub.execute_input":"2024-12-08T17:19:10.332380Z","iopub.status.idle":"2024-12-08T17:19:10.352898Z","shell.execute_reply.started":"2024-12-08T17:19:10.332331Z","shell.execute_reply":"2024-12-08T17:19:10.351794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_no_id = train_df_cleaned.drop(columns = ['id'], errors = 'ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:10.355140Z","iopub.execute_input":"2024-12-08T17:19:10.355658Z","iopub.status.idle":"2024-12-08T17:19:10.364011Z","shell.execute_reply.started":"2024-12-08T17:19:10.355607Z","shell.execute_reply":"2024-12-08T17:19:10.362538Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"markdown","source":"#### PCIAT - PCIAT","metadata":{}},{"cell_type":"code","source":"pciat_columns = [col for col in train_df_cleaned.columns if 'PCIAT-PCIAT' in col and col != 'PCIAT-PCIAT_Total']\ncorr_with_total = train_df_cleaned[pciat_columns].corrwith(train_df_cleaned['PCIAT-PCIAT_Total'])\nprint(\"Mối tương quan với PCIAT_PCIAT_TOTAL:\")\nprint(corr_with_total)\n\ntrain_df_no_id.drop(columns= pciat_columns, inplace=True)\ntrain_df_cleaned.drop(columns= pciat_columns, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:10.365748Z","iopub.execute_input":"2024-12-08T17:19:10.366121Z","iopub.status.idle":"2024-12-08T17:19:10.390104Z","shell.execute_reply.started":"2024-12-08T17:19:10.366085Z","shell.execute_reply":"2024-12-08T17:19:10.388914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Mối tương quan khá chặt chẽ","metadata":{}},{"cell_type":"markdown","source":"#### Ma trận tương quan","metadata":{}},{"cell_type":"code","source":"# Tính ma trận tương quan\ncorr_matrix = train_df_no_id.corr()\n\n# Vẽ\nplt.figure(figsize=(30, 30))\n\n# Vẽ heatmap\nplt.imshow(corr_matrix, cmap='coolwarm', interpolation='nearest', vmin=-1, vmax=1)\n\n# Thêm thanh màu (colorbar)\ncbar = plt.colorbar()\ncbar.set_label('Correlation Coefficient', fontsize=14)\n\n# Thêm nhãn \nplt.xticks(ticks=np.arange(len(corr_matrix.columns)), labels=corr_matrix.columns, rotation=90, fontsize=10)\nplt.yticks(ticks=np.arange(len(corr_matrix.index)), labels=corr_matrix.index, fontsize=10)\n\n# Thêm giá trị tương quan vào heatmap\nfor i in range(len(corr_matrix)):\n    for j in range(len(corr_matrix)):\n        plt.text(j, i, f\"{corr_matrix.iloc[i, j]:.2f}\", \n                 ha='center', va='center', color='black', fontsize=8)\n\nplt.title('Heatmap of Correlation Matrix', fontsize=30)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:10.391837Z","iopub.execute_input":"2024-12-08T17:19:10.392300Z","iopub.status.idle":"2024-12-08T17:19:17.795229Z","shell.execute_reply.started":"2024-12-08T17:19:10.392253Z","shell.execute_reply":"2024-12-08T17:19:17.794163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nếu hai cột có giá trị tương quan quá cao, để tránh lặp thông tin ta chỉ giữ lại 1 trong 2 cột. Trong trường hợp này, ta loại bỏ các cột có độ tương quan lớn hơn 0.8","metadata":{}},{"cell_type":"code","source":"threshold = 0.8\n\nto_drop = set()\nfor i in range(len(corr_matrix.columns)):\n    for j in range(i):\n        if abs(corr_matrix.iloc[i, j]) > threshold:\n            colname = corr_matrix.columns[i]\n            to_drop.add(colname)\n\nto_drop.discard('sii')\n\ntrain_df_cleaned = train_df_cleaned.drop(columns=to_drop)\n\nprint(f\"Những cột đã bị loại bỏ: {to_drop}\")\nprint(train_df_cleaned.shape)  \n\ntrain_df_cleaned = train_df_cleaned.drop(columns=['PCIAT-Season', 'PCIAT-PCIAT_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:17.796859Z","iopub.execute_input":"2024-12-08T17:19:17.797337Z","iopub.status.idle":"2024-12-08T17:19:17.862373Z","shell.execute_reply.started":"2024-12-08T17:19:17.797274Z","shell.execute_reply":"2024-12-08T17:19:17.861276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Actigraphy","metadata":{}},{"cell_type":"code","source":"actigraphy = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet')\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:17.864194Z","iopub.execute_input":"2024-12-08T17:19:17.864785Z","iopub.status.idle":"2024-12-08T17:19:17.899914Z","shell.execute_reply.started":"2024-12-08T17:19:17.864719Z","shell.execute_reply":"2024-12-08T17:19:17.898606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib.ticker import MaxNLocator\n\ndef analyze_actigraphy(id, train, only_one_week=False, small=False):\n    actigraphy = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n    \n    # Tính ngày từ time_of_day (ns)\n    time_in_days = actigraphy['relative_date_PCIAT'] + actigraphy['time_of_day'] / 86400e9\n    \n    # Lấy thông tin từ train\n    sample = train[train['id'] == id]\n    age = sample['Basic_Demos-Age'].iloc[0]\n    sex = ['boy', 'girl'][sample['Basic_Demos-Sex'].iloc[0]]\n    \n    # Thêm các cột mới\n    actigraphy['diff_seconds'] = time_in_days.diff() * 86400\n    actigraphy['norm'] = np.sqrt(actigraphy['X']**2 + actigraphy['Y']**2 + actigraphy['Z']**2)\n    \n    # Xử lý bộ lọc thời gian nếu chỉ lấy trong 1 tuần\n    if only_one_week:\n        start = np.ceil(time_in_days.min())\n        mask = (start <= time_in_days) & (time_in_days <= start + 21)\n        mask &= ~actigraphy['non-wear_flag'].astype(bool)\n    else:\n        mask = np.full(len(time_in_days), True)\n    \n    # Đặc trưng cần hiển thị\n    timelines = [('enmo', 'forestgreen'), ('light', 'orange')] if small else [\n        ('X', 'm'), ('Y', 'm'), ('Z', 'm'),\n        ('enmo', 'forestgreen'), ('anglez', 'lightblue'),\n        ('light', 'orange'), ('non-wear_flag', 'chocolate')\n    ]\n    \n    _, axs = plt.subplots(len(timelines), 1, sharex=True, figsize=(12, len(timelines) * 1.1 + 0.5))\n    \n    # Vẽ \n    for ax, (feature, color) in zip(axs, timelines):\n        ax.set_facecolor('#eeeeee')\n        ax.scatter(time_in_days[mask], actigraphy[feature][mask], color=color, label=feature, s=1)\n        ax.legend(loc='upper left', facecolor='#eeeeee')\n        if feature == 'diff_seconds':\n            ax.set_ylim(-0.5, 20.5)\n    \n    axs[-1].set_xlabel('day')\n    axs[-1].xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.tight_layout()\n    axs[0].set_title(f'id={id}, {sex}, age={age}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:17.901831Z","iopub.execute_input":"2024-12-08T17:19:17.902327Z","iopub.status.idle":"2024-12-08T17:19:17.916636Z","shell.execute_reply.started":"2024-12-08T17:19:17.902276Z","shell.execute_reply":"2024-12-08T17:19:17.915445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_actigraphy('00115b9f', train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:17.918344Z","iopub.execute_input":"2024-12-08T17:19:17.918848Z","iopub.status.idle":"2024-12-08T17:19:19.606479Z","shell.execute_reply.started":"2024-12-08T17:19:17.918798Z","shell.execute_reply":"2024-12-08T17:19:19.604964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Chỉ số ánh sáng cao cho biết những ngày này người tham gia đang ở ngoài trời hoặc tiếp xúc với ánh sáng mạnh. Chỉ số enmo cao (lớn hơn 2) cho thấy mức độ hoạt động cao.\n- Cân nhắc dùng mean, variance của enmo và light để train mô hình","metadata":{}},{"cell_type":"code","source":"analyze_actigraphy('00115b9f', train_df, small = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:19.608076Z","iopub.execute_input":"2024-12-08T17:19:19.608611Z","iopub.status.idle":"2024-12-08T17:19:20.203014Z","shell.execute_reply.started":"2024-12-08T17:19:19.608551Z","shell.execute_reply":"2024-12-08T17:19:20.201773Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"train_df_cleaned.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:20.207573Z","iopub.execute_input":"2024-12-08T17:19:20.207983Z","iopub.status.idle":"2024-12-08T17:19:20.218253Z","shell.execute_reply.started":"2024-12-08T17:19:20.207947Z","shell.execute_reply":"2024-12-08T17:19:20.216848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xác định các cột kiểu object\nobject_cols = train_df_cleaned.select_dtypes(include=['object']).columns\n\n# Thử chuyển đổi các cột kiểu object sang số (nếu được)\nfor col in object_cols:\n    try:\n        train_df_cleaned[col] = pd.to_numeric(train_df_cleaned[col])\n    except ValueError:\n        print(f\"Cột '{col}' không thể chuyển sang số và sẽ được giữ nguyên.\")\n\n# Kiểm tra lại kiểu dữ liệu sau khi chuyển đổi\nprint(train_df_cleaned.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:20.219966Z","iopub.execute_input":"2024-12-08T17:19:20.220344Z","iopub.status.idle":"2024-12-08T17:19:20.245985Z","shell.execute_reply.started":"2024-12-08T17:19:20.220307Z","shell.execute_reply":"2024-12-08T17:19:20.244315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport lightgbm as lgb\nimport numpy as np\n\n# Đặt biến mục tiêu và biến đặc trưng\ny = train_df_cleaned['sii']\nX = train_df_cleaned.drop(columns=['id', 'sii']).filter(regex='^(?!PCIAT.*$)')\n\n# Thiết lập Cross-validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)\noof = np.zeros(len(y), dtype=int)\n\n# Cross-validation loop\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # Khởi tạo và huấn luyện mô hình\n    model = lgb.LGBMClassifier(verbose=-1, random_state=1)\n    model.fit(X_train, y_train)\n    \n    # Dự đoán và tính toán điểm\n    y_pred = model.predict(X_val)\n    fold_score = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n    print(f\"Fold {fold + 1}: Score = {fold_score:.3f}\")\n    \n    # Lưu kết quả dự đoán\n    oof[val_idx] = y_pred\n\n# Đánh giá tổng thể\noverall_score = cohen_kappa_score(y, oof, weights='quadratic')\nprint(f\"\\nOverall Score: {overall_score:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:20.248046Z","iopub.execute_input":"2024-12-08T17:19:20.248877Z","iopub.status.idle":"2024-12-08T17:19:23.094618Z","shell.execute_reply.started":"2024-12-08T17:19:20.248831Z","shell.execute_reply":"2024-12-08T17:19:23.093214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Confusion matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import ConfusionMatrixDisplay\n\nConfusionMatrixDisplay.from_predictions(y, oof)\nplt.title('Confusion matrix')\nplt.xticks(np.arange(4), target_labels)\nplt.yticks(np.arange(4), target_labels)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:23.096134Z","iopub.execute_input":"2024-12-08T17:19:23.096548Z","iopub.status.idle":"2024-12-08T17:19:23.426022Z","shell.execute_reply.started":"2024-12-08T17:19:23.096492Z","shell.execute_reply":"2024-12-08T17:19:23.424689Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"season_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\n\n# Lấy các cột chứa từ \"Season\" trong tên cột\nseason_cols = [col for col in test_df.columns if 'Season' in col]\n\nfor col in season_cols:\n    # Chuyển các giá trị trong cột theo từ điển season_mapping và thay thế NaN thành giá trị 0\n    test_df[col] = test_df[col].map(season_mapping).fillna(0).astype(int)\n\n# Kiểm tra lại dữ liệu\nprint(test_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:43.445103Z","iopub.execute_input":"2024-12-08T17:19:43.445483Z","iopub.status.idle":"2024-12-08T17:19:43.472988Z","shell.execute_reply.started":"2024-12-08T17:19:43.445452Z","shell.execute_reply":"2024-12-08T17:19:43.471581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đảm bảo X_test chỉ chứa các cột giống như trong X_train\nX_test = test_df.filter(items=X.columns)\n\n# Dự đoán trên toàn bộ dữ liệu kiểm tra\ny_test_pred = model.predict(X_test)\n\n# Tạo DataFrame chứa 'id' và giá trị dự đoán của cột 'sii'\nsubmission_df = pd.DataFrame({\n    'id': test_df['id'],  # Chứa id từ dữ liệu kiểm tra\n    'sii': y_test_pred           # Dự đoán cho cột 'sii'\n})\n\n# Lưu DataFrame vào file CSV\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created: submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T17:19:23.480879Z","iopub.execute_input":"2024-12-08T17:19:23.481362Z","iopub.status.idle":"2024-12-08T17:19:23.500888Z","shell.execute_reply.started":"2024-12-08T17:19:23.481311Z","shell.execute_reply":"2024-12-08T17:19:23.499521Z"}},"outputs":[],"execution_count":null}]}