{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-28T08:41:50.929904Z","iopub.execute_input":"2024-10-28T08:41:50.930446Z","iopub.status.idle":"2024-10-28T08:41:52.052425Z","shell.execute_reply.started":"2024-10-28T08:41:50.930390Z","shell.execute_reply":"2024-10-28T08:41:52.051124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest_data = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ndata_dict = pd.read_csv(r\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T08:41:52.055044Z","iopub.execute_input":"2024-10-28T08:41:52.055958Z","iopub.status.idle":"2024-10-28T08:41:52.131022Z","shell.execute_reply.started":"2024-10-28T08:41:52.055900Z","shell.execute_reply":"2024-10-28T08:41:52.129864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.head(20))","metadata":{"execution":{"iopub.status.busy":"2024-10-28T08:41:52.132944Z","iopub.execute_input":"2024-10-28T08:41:52.133409Z","iopub.status.idle":"2024-10-28T08:41:52.165461Z","shell.execute_reply.started":"2024-10-28T08:41:52.133355Z","shell.execute_reply":"2024-10-28T08:41:52.164085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#計算資料缺失程度\nmissing_data = train_data.isnull().mean() * 100\nprint(missing_data[missing_data > 0].sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-10-28T08:41:52.168268Z","iopub.execute_input":"2024-10-28T08:41:52.168807Z","iopub.status.idle":"2024-10-28T08:41:52.187077Z","shell.execute_reply.started":"2024-10-28T08:41:52.168752Z","shell.execute_reply":"2024-10-28T08:41:52.185655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #把缺失太嚴重的欄位(attribute)去掉\ncolumns_to_drop = missing_data[missing_data > 50].index\ntrain_data = train_data.drop(columns=columns_to_drop)\n\nprint(train_data.head(20))","metadata":{"execution":{"iopub.status.busy":"2024-10-28T08:41:52.190864Z","iopub.execute_input":"2024-10-28T08:41:52.191953Z","iopub.status.idle":"2024-10-28T08:41:52.230545Z","shell.execute_reply.started":"2024-10-28T08:41:52.191907Z","shell.execute_reply":"2024-10-28T08:41:52.229039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#剩下採用插補(以下方法為目前做法，若後續因此預測出的SSI不準確需要調整可以採其他方式\n# numerical data 用 中位數插補\nnumerical_columns = [\n    'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score', 'PCIAT-PCIAT_Total', \n    'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', \n    'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'BIA-BIA_BMC', 'BIA-BIA_BMI', \n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', \n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', \n    'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', \n    'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', \n    'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nfor col in numerical_columns:\n    train_data.loc[:, col] = train_data[col].fillna(train_data[col].median())\n\n# 類別變數用眾數插補\ncategorical_columns = [\n    'Basic_Demos-Sex', 'PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Enroll_Season', 'CGAS-Season', \n    'Physical-Season', 'SDS-Season', 'PreInt_EduHx-Season', 'FGC-Season', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', \n    'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num'\n]\n\nfor col in categorical_columns:\n    train_data.loc[:, col] = train_data[col].fillna(train_data[col].mode()[0])\n\n# Drop rows where 'sii' is missing (target variable)\ntrain_data = train_data[train_data['sii'].notna()]\n\n# 檢視是否還有缺失值\nmissing_values = train_data.isnull().sum()\nprint(missing_values[missing_values > 0]) \n\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2024-10-28T08:41:52.233019Z","iopub.execute_input":"2024-10-28T08:41:52.233477Z","iopub.status.idle":"2024-10-28T08:41:52.351912Z","shell.execute_reply.started":"2024-10-28T08:41:52.233422Z","shell.execute_reply":"2024-10-28T08:41:52.350880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**清理後相比原本少了1222筆資料**","metadata":{}},{"cell_type":"code","source":"#EDA 資料內容平均、眾數、標準差\nstyled_df = train_data.describe().style.background_gradient(cmap='viridis')\nstyled_df","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:08:03.166946Z","iopub.execute_input":"2024-10-28T09:08:03.167675Z","iopub.status.idle":"2024-10-28T09:08:03.409077Z","shell.execute_reply.started":"2024-10-28T09:08:03.167623Z","shell.execute_reply":"2024-10-28T09:08:03.407770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Season**\n- for each set of measurements there is a 'season' feature which gives the season of the year when the measurements were carried out. These are the only predictive categorical features in the dataset and can be easily preprocessed.","metadata":{}},{"cell_type":"code","source":"#把season轉換成類別\ntrain_cat_columns = train_data.select_dtypes(exclude = 'number').columns\n\nfor season in train_cat_columns:\n    train_data[season] = train_data[season].fillna(0)\n    train_data[season] = train_data[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:28:20.983991Z","iopub.execute_input":"2024-10-28T09:28:20.984525Z","iopub.status.idle":"2024-10-28T09:28:20.997293Z","shell.execute_reply.started":"2024-10-28T09:28:20.984479Z","shell.execute_reply":"2024-10-28T09:28:20.995842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- there are 22 PCIAT features. These comprise answers to 20 questions (each marked out of 5), the total score and 'season' when the test was carried out.\n- The sii target is derived from the total PCIAT score:\n- 0-30 gives sii = 0\n- 31-49 gives sii = 1\n- 50-79 gives sii = 2\n- 80-100 gives sii = 3.\n","metadata":{}},{"cell_type":"markdown","source":"so we can drop all the PCIAT features from the dataset except the PCIAT Total feature which can be used as a regression target.","metadata":{}},{"cell_type":"code","source":"#cleaning the PCIAT features\nPCIAT_cols = [val for val in train_data.columns[train_data.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:24:26.235239Z","iopub.execute_input":"2024-10-28T09:24:26.236304Z","iopub.status.idle":"2024-10-28T09:24:26.244375Z","shell.execute_reply.started":"2024-10-28T09:24:26.236247Z","shell.execute_reply":"2024-10-28T09:24:26.242937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data[train_data['PCIAT-PCIAT_Total']<=30].sii.value_counts())\nprint(train_data[(train_data['PCIAT-PCIAT_Total']>30) \n    & (train_data['PCIAT-PCIAT_Total']<50)].sii.value_counts())\nprint(train_data[(train_data['PCIAT-PCIAT_Total']>=50) \n    & (train_data['PCIAT-PCIAT_Total']<80)].sii.value_counts())\nprint(train_data[train_data['PCIAT-PCIAT_Total']>=80].sii.value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:25:45.846459Z","iopub.execute_input":"2024-10-28T09:25:45.847050Z","iopub.status.idle":"2024-10-28T09:25:45.870222Z","shell.execute_reply.started":"2024-10-28T09:25:45.846993Z","shell.execute_reply":"2024-10-28T09:25:45.868877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.sii.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:26:01.066651Z","iopub.execute_input":"2024-10-28T09:26:01.067156Z","iopub.status.idle":"2024-10-28T09:26:01.077729Z","shell.execute_reply.started":"2024-10-28T09:26:01.067109Z","shell.execute_reply":"2024-10-28T09:26:01.076227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total')\ntrain_data = train_data.drop(columns = PCIAT_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T09:26:39.512833Z","iopub.execute_input":"2024-10-28T09:26:39.513490Z","iopub.status.idle":"2024-10-28T09:26:39.522682Z","shell.execute_reply.started":"2024-10-28T09:26:39.513428Z","shell.execute_reply":"2024-10-28T09:26:39.521206Z"},"trusted":true},"execution_count":null,"outputs":[]}]}