{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.10"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport numpy as np\nimport copy\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.neighbors import KNeighborsClassifier, KNeighborsRegressor\nfrom sklearn.svm import SVC, SVR\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport joblib\nfrom xgboost import XGBRegressor, XGBClassifier\nfrom sklearn.metrics import cohen_kappa_score, f1_score\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.ensemble import GradientBoostingRegressor\nimport os\nfrom sklearn.ensemble import GradientBoostingClassifier, StackingClassifier","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed_everything(2024)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# columns name and type\n# we only need type to choose the estimator\n# for filling in the Nan value\nbase_columns = [\n    # [\"id\", \"str\"],\n    [\"Basic_Demos-Age\", \"int\"],\n    [\"Basic_Demos-Sex\", \"categorical\"],\n    [\"CGAS-CGAS_Score\", \"int\"],\n    [\"Physical-Height\", \"float\"],\n    [\"Physical-Weight\", \"float\"],\n    #\n    # too much nan\n    # [\"Physical-Waist_Circumference\", \"int\"],\n    #\n    [\"Physical-Diastolic_BP\", \"int\"],\n    [\"Physical-HeartRate\", \"int\"],\n    [\"Physical-Systolic_BP\", \"int\"],\n    #\n    # too much nan\n    # [\"Fitness_Endurance-Max_Stage\", \"\"],\n    # [\"Fitness_Endurance-Time_Mins\", \"\"],\n    # [\"Fitness_Endurance-Time_Sec\", \"\"],\n    #\n    [\"FGC-FGC_CU\", \"int\"],\n    [\"FGC-FGC_CU_Zone\", \"categorical\"],\n    #\n    # too much nan\n    # [\"FGC-FGC_GSND\", \"\"],\n    # [\"FGC-FGC_GSND_Zone\", \"\"],\n    # [\"FGC-FGC_GSD\", \"\"],\n    # [\"FGC-FGC_GSD_Zone\", \"\"],\n    #\n    [\"FGC-FGC_PU\", \"int\"],\n    [\"FGC-FGC_PU_Zone\", \"categorical\"],\n    [\"FGC-FGC_SRL\", \"float\"],\n    [\"FGC-FGC_SRL_Zone\", \"categorical\"],\n    [\"FGC-FGC_SRR\", \"float\"],\n    [\"FGC-FGC_SRR_Zone\", \"categorical\"],\n    [\"FGC-FGC_TL\", \"int\"],\n    [\"FGC-FGC_TL_Zone\", \"categorical\"],\n    [\"BIA-BIA_Activity_Level_num\", \"categorical\"],\n    [\"BIA-BIA_BMC\", \"float\"],\n    [\"BIA-BIA_BMI\", \"float\"],\n    [\"BIA-BIA_BMR\", \"float\"],\n    [\"BIA-BIA_DEE\", \"float\"],\n    [\"BIA-BIA_ECW\", \"float\"],\n    [\"BIA-BIA_FFM\", \"float\"],\n    [\"BIA-BIA_FFMI\", \"float\"],\n    [\"BIA-BIA_FMI\", \"float\"],\n    [\"BIA-BIA_Fat\", \"float\"],\n    [\"BIA-BIA_Frame_num\", \"categorical\"],\n    [\"BIA-BIA_ICW\", \"float\"],\n    [\"BIA-BIA_LDM\", \"float\"],\n    [\"BIA-BIA_LST\", \"float\"],\n    [\"BIA-BIA_SMM\", \"float\"],\n    [\"BIA-BIA_TBW\", \"float\"],\n    # these two columns need to be merged\n    # so they will not be used now\n    # [\"PAQ_A-PAQ_A_Total\", \"float\"],\n    # [\"PAQ_C-PAQ_C_Total\", \"float\"],\n    [\"Activity_Summary_Score\", \"float\"],\n    #\n    # we will have to predict this ourself\n    # [\"PCIAT-PCIAT_01\", \"categorical\"],\n    # [\"PCIAT-PCIAT_02\", \"categorical\"],\n    # [\"PCIAT-PCIAT_03\", \"categorical\"],\n    # [\"PCIAT-PCIAT_04\", \"categorical\"],\n    # [\"PCIAT-PCIAT_05\", \"categorical\"],\n    # [\"PCIAT-PCIAT_06\", \"categorical\"],\n    # [\"PCIAT-PCIAT_07\", \"categorical\"],\n    # [\"PCIAT-PCIAT_08\", \"categorical\"],\n    # [\"PCIAT-PCIAT_09\", \"categorical\"],\n    # [\"PCIAT-PCIAT_10\", \"categorical\"],\n    # [\"PCIAT-PCIAT_11\", \"categorical\"],\n    # [\"PCIAT-PCIAT_12\", \"categorical\"],\n    # [\"PCIAT-PCIAT_13\", \"categorical\"],\n    # [\"PCIAT-PCIAT_14\", \"categorical\"],\n    # [\"PCIAT-PCIAT_15\", \"categorical\"],\n    # [\"PCIAT-PCIAT_16\", \"categorical\"],\n    # [\"PCIAT-PCIAT_17\", \"categorical\"],\n    # [\"PCIAT-PCIAT_18\", \"categorical\"],\n    # [\"PCIAT-PCIAT_19\", \"categorical\"],\n    # [\"PCIAT-PCIAT_20\", \"categorical\"],\n    [\"SDS-SDS_Total_T\", \"int\"],\n    [\"PreInt_EduHx-computerinternet_hoursday\", \"categorical\"],\n    # [\"sii\"],\n]\n\nbase_questions = [\n    [\"PCIAT-PCIAT_01\", \"categorical\"],\n    [\"PCIAT-PCIAT_02\", \"categorical\"],\n    [\"PCIAT-PCIAT_03\", \"categorical\"],\n    [\"PCIAT-PCIAT_04\", \"categorical\"],\n    [\"PCIAT-PCIAT_05\", \"categorical\"],\n    [\"PCIAT-PCIAT_06\", \"categorical\"],\n    [\"PCIAT-PCIAT_07\", \"categorical\"],\n    [\"PCIAT-PCIAT_08\", \"categorical\"],\n    [\"PCIAT-PCIAT_09\", \"categorical\"],\n    [\"PCIAT-PCIAT_10\", \"categorical\"],\n    [\"PCIAT-PCIAT_11\", \"categorical\"],\n    [\"PCIAT-PCIAT_12\", \"categorical\"],\n    [\"PCIAT-PCIAT_13\", \"categorical\"],\n    [\"PCIAT-PCIAT_14\", \"categorical\"],\n    [\"PCIAT-PCIAT_15\", \"categorical\"],\n    [\"PCIAT-PCIAT_16\", \"categorical\"],\n    [\"PCIAT-PCIAT_17\", \"categorical\"],\n    [\"PCIAT-PCIAT_18\", \"categorical\"],\n    [\"PCIAT-PCIAT_19\", \"categorical\"],\n    [\"PCIAT-PCIAT_20\", \"categorical\"],\n]\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LOCAL_RUN = False\n#LOCAL_RUN = True\n\nif LOCAL_RUN:\n    df_train = pd.read_csv(\"data/local/train_set.csv\")\n    df_test = pd.read_csv(\"data/local/test_set.csv\")\nelse:\n    df_train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n    df_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def merge_scores(row):\n    return (\n        row[\"PAQ_A-PAQ_A_Total\"]\n        if pd.notna(row[\"PAQ_A-PAQ_A_Total\"])\n        else row[\"PAQ_C-PAQ_C_Total\"]\n    )\ndef filter_noise(df):\n    # highest CGAS score is 100\n    df = df.drop(df[df[\"CGAS-CGAS_Score\"] > 100].index)\n    # TODO: drop all rows with too much nan\n    # convert freedom unit to metric system unit:\n    df[\"Physical-Height\"] = df[\"Physical-Height\"] * 0.0254\n    df[\"Physical-Weight\"] = df[\"Physical-Weight\"] * 0.453592\n    # Nan all Weight == 0\n    df.loc[df[\"Physical-Weight\"] <= 0, \"Physical-Weight\"] = np.nan\n    # merge PAQ_A-PAQ_A_Total and PAQ_C-PAQ_C_Total\n    df[\"Activity_Summary_Score\"] = df.apply(merge_scores, axis=1)\n    return df","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# imputation","metadata":{}},{"cell_type":"code","source":"ImpCols = [\n    # [\"id\", \"str\"],\n    \"Basic_Demos-Age\",\n    \"Basic_Demos-Sex\",\n    \"CGAS-CGAS_Score\",\n    \"Physical-Height\",\n    \"Physical-Weight\",\n    #\n    # too much nan\n    # [\"Physical-Waist_Circumference\", \"int\"],\n    #\n    \"Physical-Diastolic_BP\",\n    \"Physical-HeartRate\",\n    \"Physical-Systolic_BP\",\n    #\n    # too much nan\n    # [\"Fitness_Endurance-Max_Stage\", \"\"],\n    # [\"Fitness_Endurance-Time_Mins\", \"\"],\n    # [\"Fitness_Endurance-Time_Sec\", \"\"],\n    #\n    \"FGC-FGC_CU\",\n    \"FGC-FGC_CU_Zone\",\n    #\n    # too much nan\n    # [\"FGC-FGC_GSND\", \"\"],\n    # [\"FGC-FGC_GSND_Zone\", \"\"],\n    # [\"FGC-FGC_GSD\", \"\"],\n    # [\"FGC-FGC_GSD_Zone\", \"\"],\n    #\n    \"FGC-FGC_PU\",\n    \"FGC-FGC_PU_Zone\",\n    \"FGC-FGC_SRL\",\n    \"FGC-FGC_SRL_Zone\",\n    \"FGC-FGC_SRR\",\n    \"FGC-FGC_SRR_Zone\",\n    \"FGC-FGC_TL\",\n    \"FGC-FGC_TL_Zone\",\n    \"BIA-BIA_Activity_Level_num\",\n    \"BIA-BIA_BMC\",\n    \"BIA-BIA_BMI\",\n    \"BIA-BIA_BMR\",\n    \"BIA-BIA_DEE\",\n    \"BIA-BIA_ECW\",\n    \"BIA-BIA_FFM\",\n    \"BIA-BIA_FFMI\",\n    \"BIA-BIA_FMI\",\n    \"BIA-BIA_Fat\",\n    \"BIA-BIA_Frame_num\",\n    \"BIA-BIA_ICW\",\n    \"BIA-BIA_LDM\",\n    \"BIA-BIA_LST\",\n    \"BIA-BIA_SMM\",\n    \"BIA-BIA_TBW\",\n    #\n    # [\"PAQ_A-PAQ_A_Total\", \"float\"],\n    # [\"PAQ_C-PAQ_C_Total\", \"float\"],\n    # these two columns have been merged into this column\n    \"Activity_Summary_Score\",\n    \"SDS-SDS_Total_T\",\n    \"PreInt_EduHx-computerinternet_hoursday\",\n    # [\"sii\"],\n]\n\n\ndef impute_df(df, model, fit_mode=True, iters=5, imp_cols=ImpCols):\n    df = filter_noise(df).copy()\n\n    if fit_mode:\n        model[\"imp_cols\"] = imp_cols\n    else:\n        imp_cols=model[\"imp_cols\"]\n\n    # scale the data before imputation\n    if fit_mode:\n        imp_scaler = StandardScaler()\n        df[imp_cols] = imp_scaler.fit_transform(df[imp_cols])\n        model[\"imp_scaler\"] = imp_scaler\n\n    else:\n        df[imp_cols] = model[\"imp_scaler\"].transform(df[imp_cols])\n\n    # svr imputer\n    if fit_mode:\n        svr = SVR(kernel=\"rbf\", C=100, epsilon=0.1)\n        imputer = IterativeImputer(estimator=svr, max_iter=iters, random_state=0)\n        df[imp_cols] = imputer.fit_transform(df[imp_cols])\n        model[\"imputer\"] = imputer\n    else:\n        df[imp_cols] = model[\"imputer\"].transform(df[imp_cols])\n    \n    # unscale the scaler\n    df[imp_cols] = model[\"imp_scaler\"].inverse_transform(df[imp_cols])\n\n    return df\n\nmodel = {}\nimp_train_df = impute_df(df_train, model, fit_mode=True, iters=5)\nimp_test_df = impute_df(df_test, model, fit_mode=False)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# then predict the questionnaires","metadata":{}},{"cell_type":"code","source":"questions_cols = [\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",\n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n]\ndef predict_ques(df, model, fit_mode=False):\n\n    df = df.copy()\n    \n    if fit_mode:\n        df = df.dropna(subset=questions_cols)\n        # scale the question\n        ques_scaler = StandardScaler()\n        df[questions_cols] = ques_scaler.fit_transform(df[questions_cols])\n        model[\"ques_scaler\"] = ques_scaler\n\n        # scale the ImpCols columns\n        imp_scaler2 = StandardScaler()\n        df[ImpCols] = imp_scaler2.fit_transform(df[ImpCols])\n        model[\"imp_scaler2\"] = imp_scaler2\n    else:\n        df[ImpCols] = model[\"imp_scaler2\"].transform(df[ImpCols])\n\n    # TODO: use a different df for this\n    ques_X = df[ImpCols].copy()\n\n    total=None\n    for c in questions_cols:\n\n        if fit_mode:\n            # q_model= SVR()\n            # q_model= KNeighborsRegressor()\n            q_model= GradientBoostingRegressor()\n            q_model.fit(ques_X, df[c])\n            model[c] = q_model\n\n            pred_c = df[c]\n            # pred_c = model[c].predict(ques_X)\n\n        else:\n            pred_c = model[c].predict(ques_X)\n        \n        # pred_c = pred_c.clip(0, 5)\n        \n        total = pred_c if total is None else total + pred_c\n    \n    # scale the total value\n    # df[\"total\"] = total / len(questions_cols)\n    df[\"total\"] = total\n    \n    # then unscale the ImpCols columns\n    df[ImpCols] = model[\"imp_scaler2\"].inverse_transform(df[ImpCols])\n\n    return df\n\nq_train_df = predict_ques(imp_train_df, model, fit_mode=True)\nq_test_df = predict_ques(imp_test_df, model, fit_mode=False)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if LOCAL_RUN:\n    q_test_df[\"real_total\"] = q_test_df[questions_cols].sum(1)\n    sns.violinplot(x=q_test_df[\"sii\"], y=q_test_df[\"total\"])\n    plt.show()\n    sns.violinplot(x=q_test_df[\"sii\"], y=q_test_df[\"real_total\"])\n    plt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# then predict the sii","metadata":{}},{"cell_type":"code","source":"\ndef get_pred_df(df):\n\n    retain_columns = [\n        # \"Basic_Demos-Age\",\n        \"Basic_Demos-Sex\",\n        # \"CGAS-CGAS_Score\",\n        # # \"Physical-Height\",\n        # # \"Physical-Weight\",\n        # \"Physical-HeartRate\",\n        # \"FGC-FGC_CU\",\n        \"SDS-SDS_Total_T\",\n        # \"PreInt_EduHx-computerinternet_hoursday\",\n        # \"Activity_Summary_Score\",\n        \"total\"\n    ]\n    # df for predition\n    pred_df = df[retain_columns].copy()\n    # calculate BMI\n    pred_df[\"Physical-BMI\"] = df[\"Physical-Weight\"] / (df[\"Physical-Height\"] ** 2)\n    # PCIAT_cols = [c for c, t in base_columns if c.startswith(\"PCIAT-PCIAT\")]\n    # pred_df[\"PCIAT-PCIAT_Total\"] = df[PCIAT_cols].sum(axis=1)\n    return pred_df\n\ndef predict_sii(df, model, fit_mode=False):\n\n    if fit_mode:\n        df = df.dropna(subset=[\"sii\"])\n\n    pred_df = get_pred_df(df).copy()\n\n    # then scale the pred_df\n    if fit_mode:\n        pred_df_scaler = StandardScaler()\n        pred_df[pred_df.columns.tolist()] = pred_df_scaler.fit_transform(pred_df[pred_df.columns.tolist()])\n        model[\"pred_df_scaler\"] = pred_df_scaler\n    else:\n        pred_df[pred_df.columns.tolist()] = model[\"pred_df_scaler\"].fit_transform(pred_df[pred_df.columns.tolist()])\n\n    # pred_df[\"total\"].apply(lambda x: x*10)\n\n    if fit_mode:\n        y = df[\"sii\"]\n        clf = KNeighborsClassifier(n_neighbors=6, )\n        # clf = SVC(decision_function_shape='ovr', class_weight=\"balanced\")\n        # clf = GradientBoostingClassifier()\n\n        clf.fit(pred_df, y)\n        model[\"clf\"] = clf\n    else:\n        y_pred = model[\"clf\"].predict(pred_df)\n        return y_pred\n\n    return None\n\npredict_sii(q_train_df, model, fit_mode=True)\npred_y = predict_sii(q_test_df, model, fit_mode=False)\n\ndf_sub = df_test.copy()\ndf_sub[\"sii\"] = pred_y\ndf_sub[[\"id\", \"sii\"]].to_csv('submission.csv', index=False)\n\nif LOCAL_RUN:\n    y_test = df_test[\"sii\"]\n    print(cohen_kappa_score(pred_y, y_test))\n    print(f1_score(pred_y, y_test, average=\"weighted\"))\n\n    # current highest is 0.26","metadata":{},"outputs":[],"execution_count":null}]}