{"metadata":{"kernelspec":{"display_name":"base","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.8.8"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nfrom concurrent.futures import ThreadPoolExecutor\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import KFold\n\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nfrom scipy.stats import pointbiserialr\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parquet_train = \"./data/series_train.parquet\"\ntrain_df = pd.read_csv(\"./data/train.csv\")\n\nparquet_test = \"./data/series_test.parquet\"\ntest_df = pd.read_csv(\"./data/test.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(missing_values=np.nan, strategy='mean')\nlr = LinearRegression()\nrf = RandomForestClassifier(n_estimators=100, random_state=42)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_parquet(file_name,parquet_dir):\n    \n    user_id = file_name.split('=')[1]\n    df = pd.read_parquet(os.path.join(parquet_dir, file_name))\n    df = df[df['non-wear_flag'] <= 0.5]\n    df = df.drop(columns=['step', 'time_of_day', 'weekday', 'quarter', 'battery_voltage', 'non-wear_flag', 'relative_date_PCIAT'])\n    df_mean = df.mean().to_frame().T\n    df_mean['id'] = user_id\n    \n    return df_mean\n\ndef prepare_data(parquet_dir,df):\n\n    parquet_files = os.listdir(parquet_dir)\n    acc_stats = []\n    \n    with ThreadPoolExecutor() as executor:\n        \n        results = executor.map(lambda file_name: process_parquet(file_name, parquet_dir), parquet_files)\n       \n        acc_stats.extend(results)\n\n        df_acc_stats = pd.concat(acc_stats, ignore_index=True)\n        acc_cols = df_acc_stats.drop('id',axis=1).columns\n        df = pd.merge(df, df_acc_stats, on='id',how='left')\n        df[acc_cols] = df[acc_cols].fillna(0)\n    return df\n\n\ndef impute_missing_values(df, independent_cols, target, estimator):\n\n    x_train = df.dropna(subset=[target])[independent_cols]\n    y_train = df[target].dropna()\n    x_test = df[df[target].isna()][independent_cols]\n    \n    pipeline = Pipeline(steps=[('model', estimator)])\n    pipeline.fit(x_train, y_train)\n\n    x_test['predictions'] = pipeline.predict(x_test)\n    \n    return x_test['predictions']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = prepare_data(parquet_train,train_df)\ntest_df = prepare_data(parquet_test,test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pre_processing(df):\n\n    # Drop all season columns as they dont add much information\n    df = df.loc[:, ~df.columns.str.contains('Season')]\n    df = df.drop('Physical-Waist_Circumference',axis=1,errors='ignore')\n\n    # Convert Height in inches to meter squared\n    df['Physical-Height'] = df['Physical-Height'] / 39.37\n    df['Physical-Height'] = df['Physical-Height'] * df['Physical-Height']\n\n    # Convert weight in lbs to kg\n    df['Physical-Weight'] = df['Physical-Weight'] / 2.205\n\n    df['Physical-Height'] = df['Physical-Height'].fillna(impute_missing_values(df = df,independent_cols = ['Basic_Demos-Age', 'Basic_Demos-Sex'],target = 'Physical-Height',estimator = lr))\n    df['Physical-Weight'] = df['Physical-Weight'].fillna(impute_missing_values(df = df,independent_cols=['Basic_Demos-Age', 'Basic_Demos-Sex'],target ='Physical-Weight',estimator=lr))\n\n    df['Physical-BMI'] = df['Physical-BMI'].fillna(df['Physical-Weight'] / df['Physical-Height'])\n\n    # fill other missing values\n    df['Physical-Diastolic_BP'] = df['Physical-Diastolic_BP'].fillna(impute_missing_values(df,['Physical-BMI','Physical-Height','Physical-Weight','Basic_Demos-Age'],'Physical-Diastolic_BP',lr))\n    df['Physical-Systolic_BP'] = df['Physical-Systolic_BP'].fillna(impute_missing_values(df,['Physical-BMI','Physical-Height','Physical-Weight','Basic_Demos-Age','Physical-Diastolic_BP'],'Physical-Systolic_BP',lr))\n    df['Physical-HeartRate']=df['Physical-HeartRate'].fillna(impute_missing_values(df,['Basic_Demos-Age','Basic_Demos-Sex','Physical-BMI','Physical-Height','Physical-Weight','Physical-Diastolic_BP',\"Physical-Systolic_BP\"],\"Physical-HeartRate\",lr))\n\n\n    df = df.drop(['Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec'],axis=1)\n    df['Fitness_Endurance-Max_Stage'] = df['Fitness_Endurance-Max_Stage'].fillna(impute_missing_values(df,['Basic_Demos-Age','Basic_Demos-Sex','Physical-BMI','Physical-Height','Physical-Weight','Physical-Diastolic_BP',\"Physical-Systolic_BP\",\"Physical-HeartRate\"],'Fitness_Endurance-Max_Stage',lr))\n\n    mask = (df.isnull().sum() / len(df)) * 100\n\n    # print(mask[mask>70].index)\n    to_delete = mask[mask>70].index\n    df = df.drop(to_delete,axis=1,errors='ignore')\n\n\n    temp = df.dropna(subset=['FGC-FGC_CU','FGC-FGC_CU_Zone'])[['FGC-FGC_CU','FGC-FGC_CU_Zone']]\n    #corr, p_value = pointbiserialr(temp['FGC-FGC_CU'], temp['FGC-FGC_CU_Zone'])\n    #print(corr,p_value)\n\n    df = df.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'],axis=1,errors='ignore')\n\n    independent_cols = ['Basic_Demos-Age','Basic_Demos-Sex','Physical-BMI','Physical-Height','Physical-Weight','Physical-Diastolic_BP',\"Physical-Systolic_BP\",\"Physical-HeartRate\",\"Fitness_Endurance-Max_Stage\"]\n    df['FGC-FGC_CU'] = df['FGC-FGC_CU'].fillna(impute_missing_values(df,independent_cols,'FGC-FGC_CU',lr))\n    df['FGC-FGC_PU'] = df['FGC-FGC_PU'].fillna(impute_missing_values(df,independent_cols,'FGC-FGC_PU',lr))\n    df['FGC-FGC_SRR'] = df['FGC-FGC_SRR'].fillna(impute_missing_values(df,independent_cols,'FGC-FGC_SRR',lr))\n    df['FGC-FGC_SRL'] = df['FGC-FGC_SRL'].fillna(impute_missing_values(df,independent_cols,'FGC-FGC_SRL',lr))\n    df['FGC-FGC_TL'] = df['FGC-FGC_TL'].fillna(impute_missing_values(df,independent_cols,'FGC-FGC_TL',lr))\n\n\n    df = df.drop('BIA-BIA_BMI',axis=1,errors='ignore')  \n\n    # print(df[[\"BIA-BIA_FFM\",\"BIA-BIA_FFMI\",\"BIA-BIA_FMI\",\"BIA-BIA_Fat\"]].corr())\n    df = df.drop(['BIA-BIA_Fat','BIA-BIA_FFMI','BIA-BIA_FMI','BIA-BIA_Frame_num'],errors='ignore',axis=1)\n    \n    # print(df[[\"BIA-BIA_ECW\",\"BIA-BIA_ICW\",\"BIA-BIA_TBW\"]].corr())\n    df = df.drop(['BIA-BIA_ICW',\"BIA-BIA_TBW\"],axis=1,errors='ignore')\n\n    # print(df[[\"BIA-BIA_LDM\",\"BIA-BIA_LST\",\"BIA-BIA_SMM\"]].corr())\n    df = df.drop(['BIA-BIA_LST',\"BIA-BIA_SMM\"],axis=1,errors='ignore')\n\n    # print(df.filter(like=\"BIA_\").corr())\n    df = df.drop(['BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM','BIA-BIA_LDM'],errors='ignore',axis=1)\n\n\n    df[['CGAS-CGAS_Score','BIA-BIA_BMC','PAQ_C-PAQ_C_Total','SDS-SDS_Total_Raw']] = imputer.fit_transform(df[['CGAS-CGAS_Score','BIA-BIA_BMC','PAQ_C-PAQ_C_Total','SDS-SDS_Total_Raw']])\n    df['CGAS-CGAS_Score'] = df['CGAS-CGAS_Score'].astype(int)\n    df['SDS-SDS_Total_Raw'] = df['SDS-SDS_Total_Raw'].astype(int)\n\n    df = df.drop('SDS-SDS_Total_T',axis=1,errors='ignore')\n    df = df.drop(df.filter(like = 'PCIAT-PCIAT_').iloc[:,:-1].columns,axis=1,errors='ignore')\n\n    df['BIA-BIA_Activity_Level_num'] = df['BIA-BIA_Activity_Level_num'].fillna(impute_missing_values(df,['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'],'BIA-BIA_Activity_Level_num',rf))\n    df['PreInt_EduHx-computerinternet_hoursday'] = df['PreInt_EduHx-computerinternet_hoursday'].fillna(impute_missing_values(df,['SDS-SDS_Total_Raw'],'PreInt_EduHx-computerinternet_hoursday',rf))\n\n    try:\n        df = df.dropna(subset=['PCIAT-PCIAT_Total'])\n    except:\n        ...\n\n    return df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pre_processing(train_df)\ntest_df = pre_processing(test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n    'random_state': 42,\n    'n_estimators': 300\n}\n\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': 42,\n    'tree_method': 'exact'\n}\n\ncb_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\nlgb = LGBMRegressor(**lgb_params)\nxgb = XGBRegressor(**xgb_params)\ncb = CatBoostRegressor(**cb_params)\n\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', lgb),\n    ('xgboost', xgb),\n    ('catboost', cb)\n])\n\ndef cohen_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\ndef train(df):\n    \n    X = df.drop(['id', 'PCIAT-PCIAT_Total', 'sii'], axis=1)\n    y = df['PCIAT-PCIAT_Total']\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    \n    kc_score = []\n    for train_idx, val_idx in kf.split(X):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        y_val_cat = train_df.loc[y_val.index, 'sii']\n        \n        voting_model.fit(X_train, y_train)\n        \n        y_val_pred = voting_model.predict(X_val)\n        y_val_pred_cat = np.select([y_val_pred <= 30, y_val_pred <= 49, y_val_pred <= 79],[0.0, 1.0, 2.0],default=3.0)\n        \n        kc_score.append(cohen_kappa(y_val_cat, y_val_pred_cat))\n        \n    print(f'Cohen Kappa Score: {np.mean(kc_score):.4f}')\n    \n    return voting_model\n\nvoting_model = train(train_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['PCIAT-PCIAT_Total'] = voting_model.predict(test_df.drop('id', axis=1))\n\ntest_df['sii'] = np.select(\n    [test_df['PCIAT-PCIAT_Total'] <= 30, \n     test_df['PCIAT-PCIAT_Total'] <= 49, \n     test_df['PCIAT-PCIAT_Total'] <= 79],\n    [0.0, 1.0, 2.0],\n    default=3.0)\n\ntest_df['sii']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[['id','sii']].to_csv('submission.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}