{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":10184847,"sourceType":"datasetVersion","datasetId":6291863}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:22.367309Z","iopub.execute_input":"2024-12-19T09:31:22.367729Z","iopub.status.idle":"2024-12-19T09:31:22.374116Z","shell.execute_reply.started":"2024-12-19T09:31:22.367692Z","shell.execute_reply":"2024-12-19T09:31:22.372886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Not cleaned, but this was best submitted kernel.\n\n- Iterative catboost imputer\n- 15 seed 5 fold CV regression and classification\n- thresholding 0 and 3 class of classification\n- Regression using total pciat and default thresholds.\n- Parquet used, and many features eliminated.\n\nCause of poor score likely due to large feature elimination or not using sii and optimized QWK. Balanced overfitting well, public score close to private score, but score was just not good enough.","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print('Mem. usage decreased to {:5.2f} Mb ({:.1f}%reduction)'.format(\n            end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:22.573941Z","iopub.execute_input":"2024-12-19T09:31:22.574367Z","iopub.status.idle":"2024-12-19T09:31:22.587512Z","shell.execute_reply.started":"2024-12-19T09:31:22.574329Z","shell.execute_reply":"2024-12-19T09:31:22.586123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport plotly.express as px\nimport gc\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, KFold\nfrom sklearn.metrics import roc_auc_score, accuracy_score, mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder, power_transform\nfrom sklearn.decomposition import PCA\nfrom sklearn.utils import class_weight\nimport shutil\nimport os\nimport glob\nimport pathlib\nimport seaborn as sns\nfrom catboost import CatBoostClassifier, CatBoostRegressor\nfrom tqdm import tqdm\nimport optuna\nfrom lightgbm import LGBMClassifier, LGBMRegressor\nimport itertools\nfrom collections import Counter\nfrom sklearn.metrics import cohen_kappa_score\nfrom functools import partial\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nimport random\nfrom sklearn.utils.class_weight import compute_class_weight\nimport scipy\nfrom scipy.stats import mode\nfrom numba.core.errors import NumbaDeprecationWarning, NumbaPendingDeprecationWarning\nimport warnings\nfrom scipy.optimize import minimize\n\nimport tensorflow as tf\nimport dask.dataframe as dd\nfrom dask import delayed\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport pyarrow.parquet as pq\n\nwarnings.simplefilter('ignore', category = NumbaDeprecationWarning)\nwarnings.simplefilter('ignore', category = NumbaPendingDeprecationWarning)\nwarnings.simplefilter('ignore', category = UserWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:22.749782Z","iopub.execute_input":"2024-12-19T09:31:22.750268Z","iopub.status.idle":"2024-12-19T09:31:44.133929Z","shell.execute_reply.started":"2024-12-19T09:31:22.750207Z","shell.execute_reply":"2024-12-19T09:31:44.132408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndict_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n\ntrain_df.columns = [i.lower() for i in train_df.columns]\ntest_df.columns = [i.lower() for i in test_df.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.136328Z","iopub.execute_input":"2024-12-19T09:31:44.137291Z","iopub.status.idle":"2024-12-19T09:31:44.238949Z","shell.execute_reply.started":"2024-12-19T09:31:44.137243Z","shell.execute_reply":"2024-12-19T09:31:44.237411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CatKappa(object):\n    def evaluate(self, approxes, targets, weights):\n        gt = targets\n        gt = gt.astype('int32')\n        pred = approxes[0]\n        if max(gt) > 10:\n            gt = np.where(gt < 31, 0, np.where(gt < 50, 1, np.where(gt < 80, 2, 3)))\n            pred = np.where(pred < 31, 0, np.where(pred < 50, 1, np.where(pred < 80, 2, 3)))\n        else:\n            gt = np.clip(np.round(gt), 0, 4).astype(int)\n            pred = np.clip(np.round(pred), 0, 4).astype(int)\n        kap = cohen_kappa_score(gt, pred, labels = [0, 1, 2, 3], weights = 'quadratic')\n        if weights is not None:\n            weighted_kap = kap * np.sum(weights)\n            weight_sum = np.sum(weights)\n        else:\n            weighted_kap = kap\n            weight_sum = len(targets)        \n        return kap, weight_sum\n    def is_max_optimal(self):\n        return True\n    def get_final_error(self, error, weight):\n        return error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.240514Z","iopub.execute_input":"2024-12-19T09:31:44.240895Z","iopub.status.idle":"2024-12-19T09:31:44.253191Z","shell.execute_reply.started":"2024-12-19T09:31:44.240856Z","shell.execute_reply":"2024-12-19T09:31:44.251420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def xgb_acc(y_true, y_pred):\n    acc = (y_pred == y_true).sum() / len(y_true)\n    return -acc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.256617Z","iopub.execute_input":"2024-12-19T09:31:44.257107Z","iopub.status.idle":"2024-12-19T09:31:44.274012Z","shell.execute_reply.started":"2024-12-19T09:31:44.257066Z","shell.execute_reply":"2024-12-19T09:31:44.272415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def kappa(gt, pred):\n\n    if max(pred) > 10:\n        pred = np.where(pred < 31, 0, np.where(pred < 50, 1, np.where(pred < 80, 2, 3)))\n    if max(gt) > 10:\n        gt = np.where(gt < 31, 0, np.where(gt < 50, 1, np.where(gt < 80, 2, 3)))\n    else:\n        gt = np.clip(np.round(gt), 0, 4).astype(int)\n        pred = np.clip(np.round(pred), 0, 4).astype(int)\n        \n    kap = cohen_kappa_score(gt, pred, labels = [0, 1, 2, 3], weights = 'quadratic')\n    return kap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.275975Z","iopub.execute_input":"2024-12-19T09:31:44.276497Z","iopub.status.idle":"2024-12-19T09:31:44.294560Z","shell.execute_reply.started":"2024-12-19T09:31:44.276447Z","shell.execute_reply":"2024-12-19T09:31:44.293293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def xgb_kappa(y_true, y_pred):\n    if max(y_true) > 10:\n        y_pred = np.where(y_pred < 31, 0, np.where(y_pred < 50, 1, np.where(y_pred < 80, 2, 3)))\n        y_true = np.where(y_true < 31, 0, np.where(y_true < 50, 1, np.where(y_true < 80, 2, 3)))\n    else:\n        y_pred = np.clip(np.round(y_pred), 0, 4).astype(int)\n        y_true = np.clip(np.round(y_true), 0, 4).astype(int)\n        \n    kap =  - cohen_kappa_score(y_true, y_pred, labels = [0, 1, 2, 3], weights = 'quadratic')\n    return kap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.296247Z","iopub.execute_input":"2024-12-19T09:31:44.296614Z","iopub.status.idle":"2024-12-19T09:31:44.311400Z","shell.execute_reply.started":"2024-12-19T09:31:44.296580Z","shell.execute_reply":"2024-12-19T09:31:44.310230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def light_kappa(y_true, pred):\n    if pred.shape[-1] == 4:\n        pred = pred.reshape(len(y_true), -1)\n        pred = np.argmax(pred, axis=1)\n    kap = kappa(y_true, pred)\n    return 'kap', kap, True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.313679Z","iopub.execute_input":"2024-12-19T09:31:44.314296Z","iopub.status.idle":"2024-12-19T09:31:44.328114Z","shell.execute_reply.started":"2024-12-19T09:31:44.314240Z","shell.execute_reply":"2024-12-19T09:31:44.326657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"id_list = [i.split('=')[1] for i in os.listdir('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')]\nidle_list = []\nnon_idle = []\ndef check_idle_status(i):\n    df = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={i}/part-0.parquet')\n    if df['relative_date_PCIAT'].value_counts().iloc[0] == 17280:\n        return i, 'non_idle'\n    else:\n        return i, 'idle'\nwith ProcessPoolExecutor() as executor:\n    results = list(tqdm(executor.map(check_idle_status, id_list), total=len(id_list)))\n    \nfor i, status in results:\n    if status == 'non_idle':\n        non_idle.append(i)\n    else:\n        idle_list.append(i)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:31:44.330179Z","iopub.execute_input":"2024-12-19T09:31:44.330672Z","iopub.status.idle":"2024-12-19T09:32:14.551695Z","shell.execute_reply.started":"2024-12-19T09:31:44.330619Z","shell.execute_reply":"2024-12-19T09:32:14.550302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_id_list = [i.split('=')[1] for i in os.listdir('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')]\ntest_idle = []\ntest_non_idle = []\ndef check_idle_test(i):\n    df = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/id={i}/part-0.parquet')\n    if df['relative_date_PCIAT'].value_counts().iloc[0] == 17280:\n        return i, 'non_idle'\n    else:\n        return i, 'idle'\nwith ProcessPoolExecutor() as executor:\n    results = list(tqdm(executor.map(check_idle_test, test_id_list), total = len(test_id_list)))\nfor i, status in results:\n    if status == 'non_idle':\n        test_non_idle.append(i)\n    else:\n        test_idle.append(i)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:14.553396Z","iopub.execute_input":"2024-12-19T09:32:14.553782Z","iopub.status.idle":"2024-12-19T09:32:15.159876Z","shell.execute_reply.started":"2024-12-19T09:32:14.553737Z","shell.execute_reply":"2024-12-19T09:32:15.158460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nTRAIN_LIST = [p.split('=')[1] for p in os.listdir(TRAIN_PATH)]\n\nTEST_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nTEST_LIST = [p.split('=')[1] for p in os.listdir(TEST_PATH)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.163399Z","iopub.execute_input":"2024-12-19T09:32:15.163803Z","iopub.status.idle":"2024-12-19T09:32:15.172765Z","shell.execute_reply.started":"2024-12-19T09:32:15.163762Z","shell.execute_reply":"2024-12-19T09:32:15.171206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_train = train_df[~train_df.sii.isna()].reset_index(drop = True)\ntraining_, validing_ = train_test_split(full_train, shuffle = True, random_state = 42)\ntraining_ = training_.reset_index(drop = True)\nvaliding_ = validing_.reset_index(drop = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.174979Z","iopub.execute_input":"2024-12-19T09:32:15.175502Z","iopub.status.idle":"2024-12-19T09:32:15.206891Z","shell.execute_reply.started":"2024-12-19T09:32:15.175447Z","shell.execute_reply":"2024-12-19T09:32:15.204858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def season_looper(val):\n    seasons = [0, 1, 2, 3]\n    if pd.isna(val['physical-season']) or pd.isna(val['basic_demos-enroll_season']):\n        return -1\n    enroll_index = seasons.index(val['basic_demos-enroll_season'])\n    physical_index = seasons.index(val['physical-season'])\n    if enroll_index == physical_index:\n        return 0\n    elif enroll_index < physical_index:\n        return physical_index - enroll_index\n    else:\n        return (physical_index - enroll_index) % 4\ndef cgas_looper(val):\n    seasons = [0, 1, 2, 3]\n    if pd.isna(val['physical-season']) or pd.isna(val['cgas-season']):\n        return -1\n    enroll_index = seasons.index(val['physical-season'])\n    physical_index = seasons.index(val['cgas-season'])\n    if enroll_index == physical_index:\n        return 0\n    elif enroll_index < physical_index:\n        return physical_index - enroll_index\n    else:\n        return (physical_index - enroll_index) % 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.208797Z","iopub.execute_input":"2024-12-19T09:32:15.209389Z","iopub.status.idle":"2024-12-19T09:32:15.219316Z","shell.execute_reply.started":"2024-12-19T09:32:15.209345Z","shell.execute_reply":"2024-12-19T09:32:15.217931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def imputer_features(df, parq_train = True):\n    if parq_train is True:\n        df['has_parquett'] = np.where(df.id.isin(TRAIN_LIST), 1, 0)\n        df['idlefeat'] = np.where(df.id.isin(idle_list), 1, 0)\n        df['nonidlefeat'] = np.where(df.id.isin(non_idle), 1, 0)\n    else:\n        df['has_parquett'] = np.where(df.id.isin(TEST_LIST), 1, 0)\n        df['idlefeat'] = np.where(df.id.isin(test_idle), 1, 0)\n        df['nonidlefeat'] = np.where(df.id.isin(test_non_idle), 1, 0)\n    season_mapper = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3}\n    for i in [p for p in df.columns if 'season' in p]:\n        df[i] = df[i].map(season_mapper)\n    df = df.drop([i for i in df.columns if 'zone' in i], axis = 1)\n    feat_idx = df[~(df['paq_c-paq_c_total'].isna()) & (df['basic_demos-age'] < 8) | (df['basic_demos-age'] > 14)].index.to_list() +\\\n    df[~(df['sds-season'].isna()) & ((df['basic_demos-age'] < 6) | (df['basic_demos-age'] > 15))].index.to_list() +\\\n    df[~(df['paq_a-paq_a_total'].isna()) & (df['basic_demos-age'] < 14) | (df['basic_demos-age'] > 18)].index.to_list() +\\\n    df[~(df['fitness_endurance-season'].isna()) & (df['basic_demos-age'] > 12)].index.to_list() +\\\n    df[~(df['fgc-fgc_gsnd'].isna()) & (df['basic_demos-age'] < 10) | (df['basic_demos-age'] > 18)].index.to_list() +\\\n    df[~(df['fgc-fgc_gsd'].isna()) & (df['basic_demos-age'] < 10) | (df['basic_demos-age'] > 18)].index.to_list() +\\\n    df[df['bia-bia_fat'] < 2].index.to_list() +\\\n    df[df['bia-bia_fmi'] < 0].index.to_list() +\\\n    df[df['physical-bmi'] <= 0].index.to_list() +\\\n    df[df['physical-weight'] < 10].index.to_list() +\\\n    df[df['physical-systolic_bp'] < df['physical-diastolic_bp']].index.to_list() +\\\n    df[df['physical-diastolic_bp'] < 45].index.to_list() +\\\n    df[df['physical-systolic_bp'] < 75].index.to_list() +\\\n    df[df['physical-heartrate'] < 50].index.to_list()\n    df.loc[sorted(list(set(feat_idx))), 'has_strange'] = 1\n    df.has_strange = df.has_strange.fillna(0).astype(int)\n    df['physical_sea_isna'] = df['physical-season'].isna().astype(int)\n    df['cgas_sea_isna'] = df['cgas-season'].isna().astype(int)\n    df['cgas_score_isna'] = df['cgas-cgas_score'].isna().astype(int)\n    df['bmi_isna'] = df['physical-bmi'].isna().astype(int)\n    df['height_isna'] = df['physical-height'].isna().astype(int)\n    df['weight_isna'] = df['physical-weight'].isna().astype(int)\n    df['waist_cir_isna'] = df['physical-waist_circumference'].isna().astype(int)\n    df['diastolic_isna'] = df['physical-diastolic_bp'].isna().astype(int)\n    df['heartrate_isna'] = df['physical-heartrate'].isna().astype(int)\n    df['systolic_isna'] = df['physical-systolic_bp'].isna().astype(int)\n    df['fgc_sea_isna'] = df['fgc-season'].isna().astype(int)\n    df['fitness_endmax_isna'] = df['fitness_endurance-max_stage'].isna().astype(int)\n    df['fitness_endsea_isna'] = df['fitness_endurance-season'].isna().astype(int)\n    df['cu_isna'] = df['fgc-fgc_cu'].isna().astype(int)\n    df['pu_isna'] = df['fgc-fgc_pu'].isna().astype(int)\n    df['srl_isna'] = df['fgc-fgc_srl'].isna().astype(int)\n    df['srr_isna'] = df['fgc-fgc_srr'].isna().astype(int)\n    df['tl_isna'] = df['fgc-fgc_tl'].isna().astype(int)\n    df['gsnd_isna'] = df['fgc-fgc_gsnd'].isna().astype(int)\n    df['gsd_isna'] = df['fgc-fgc_gsd'].isna().astype(int)\n    df['cu_zero'] = np.where(df['fgc-fgc_cu'] == 0, 1, 0)\n    df['pu_zero'] = np.where(df['fgc-fgc_pu'] == 0, 1, 0)\n    df['paqa_isna'] = df['paq_a-season'].isna().astype(int)\n    df['paqc_isna'] = df['paq_c-season'].isna().astype(int)\n    df['bia_isna'] = df['bia-season'].isna().astype(int)\n    df['bia_fat_isna'] = df['bia-bia_fat'].isna().astype(int)\n    df['sds_isna'] = df['sds-season'].isna().astype(int)\n    df['preint_isna'] = df['preint_eduhx-season'].isna().astype(int)\n    df['preint_hour_isna'] = df['preint_eduhx-computerinternet_hoursday'].isna().astype(int)\n    df['sds_cor'] = np.where(df['sds-season'].isna() & (df['basic_demos-age'] >= 6) & (df['basic_demos-age'] <= 15), 1, 0)\n    df['sds_outrange'] = np.where(~(df['sds-season'].isna()) & ((df['basic_demos-age'] < 6) | (df['basic_demos-age'] > 15)), 1, 0)\n    df['over18'] = (df['basic_demos-age'] > 18).astype(int)\n    df['physical_sea_diff'] = df.apply(lambda x: season_looper(x), axis = 1)\n    df['cgas_sea_diff'] = df.apply(lambda x: cgas_looper(x), axis = 1)\n    # sort weird/ impossible values\n    df.loc[df['bia-bia_fat'] < 2, 'bia-bia_fat'] = np.nan\n    df.loc[df['bia-bia_fmi'] < 0, 'bia-bia_fmi'] = np.nan\n    df.loc[df['physical-bmi'] <= 0, 'physical-bmi'] = np.nan\n    df.loc[df['physical-weight'] < 10, 'physical-weight'] = np.nan\n    df.loc[df['physical-systolic_bp'] < df['physical-diastolic_bp'], ['physical-systolic_bp', 'physical-diastolic_bp']] = np.nan\n    df.loc[df['physical-diastolic_bp'] < 45, 'physical-diastolic_bp'] = np.nan\n    df.loc[df['physical-systolic_bp'] < 75, 'physical-systolic_bp'] = np.nan\n    df.loc[df['physical-heartrate'] < 50, 'physical-heartrate'] = np.nan\n    df = df.drop(['paq_c-season', 'paq_a-season', 'fitness_endurance-season', 'fitness_endurance-time_sec',\n                   'sds-sds_total_t', 'bia-bia_bmi', 'bia-bia_ffm'], axis = 1)\n    df = df.drop([p for p in df.columns if 'season' in p and 'basic_demos-' not in p], axis = 1)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.221257Z","iopub.execute_input":"2024-12-19T09:32:15.221737Z","iopub.status.idle":"2024-12-19T09:32:15.253015Z","shell.execute_reply.started":"2024-12-19T09:32:15.221694Z","shell.execute_reply":"2024-12-19T09:32:15.251476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def non_idle_features(df):\n    df = df.copy()\n    df.loc[:, 'anglez'] = df.anglez.abs()\n    features = ['enmo', 'light', 'anglez']\n    next_dfs = []\n    temp_df = df.copy()\n    for p in temp_df['relative_date_PCIAT'].unique():\n        vc = temp_df[temp_df['relative_date_PCIAT'] == p]['non-wear_flag'].value_counts()\n        if 1 in vc and 0 in vc:\n            if vc[0] / 17280 > 0.35:\n                next_dfs.append(temp_df[temp_df['relative_date_PCIAT'] == p])\n        elif 0 in vc and 1 not in vc:\n            next_dfs.append(temp_df[temp_df['relative_date_PCIAT'] == p])\n        else:\n            continue\n    if len(next_dfs) == 0:\n        return None\n    df = pd.concat(next_dfs)\n    df = df[df['non-wear_flag'] == 0]\n    for p in features:\n        df[f'mean_{p}'] = df[p].mean()\n        df[f'std_{p}'] = df[p].std()\n    cols = [i for i in df.columns if 'std_' in i] + [i for i in df.columns if 'mean_' in i] + ['light']\n    df = df[cols]\n    df = df.mean()\n    df = pd.DataFrame(df).T\n    if df.empty:\n        return None\n    elif np.isnan(df['light'].iloc[0]):\n        return None\n    return df.drop('light', axis = 1)\ndef non_idle_model(training52, ids_, data_type = 'train'):\n    def feat_do(id__):\n        if data_type == 'train':\n            file_path = f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id__}/part-0.parquet'\n        else:\n            file_path = f'/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/id={id__}/part-0.parquet'\n        df = pd.read_parquet(file_path)\n        df = non_idle_features(df)\n        temp_df = training52.copy()[training52.id == id__]\n        if df is not None:\n            df.loc[:, temp_df.columns] = temp_df.values[0]\n            df['id'] = id__\n            return df\n        return None\n    with ThreadPoolExecutor(max_workers = 4) as executor:\n        results = list(tqdm(executor.map(feat_do, ids_)))\n    trainingx = reduce_mem_usage(pd.concat(results, ignore_index = True), verbose = False)\n    del results\n    gc.collect()\n    return trainingx\ngc.collect()\n\ntra_parq_id = [i for i in non_idle if i in training_.id.values]\nval_parq_id = [i for i in non_idle if i in validing_.id.values]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.254593Z","iopub.execute_input":"2024-12-19T09:32:15.254992Z","iopub.status.idle":"2024-12-19T09:32:15.659349Z","shell.execute_reply.started":"2024-12-19T09:32:15.254950Z","shell.execute_reply":"2024-12-19T09:32:15.657997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_feat = imputer_features(full_train[test_df.columns].copy())\ntrain_feat = imputer_features(training_[test_df.columns].copy())\nvalid_feat = imputer_features(validing_[test_df.columns].copy())\ntest_feat = imputer_features(test_df.copy())\n\ntes_parq = non_idle_model(full_feat.copy(), tra_parq_id + val_parq_id, data_type = 'train')\nfull_parq = full_feat.copy().merge(tes_parq.iloc[:, :7], how = 'left', on = 'id')\ntrain_parq = train_feat.copy().merge(tes_parq[tes_parq.id.isin(train_feat.id)].iloc[:, :7], how = 'left', on = 'id')\nvalid_parq = valid_feat.copy().merge(tes_parq[tes_parq.id.isin(valid_feat.id)].iloc[:, :7], how = 'left', on = 'id')\ntes_parq = non_idle_model(test_feat.copy(), test_non_idle, data_type = 'test')\ntest_parq = test_feat.copy().merge(tes_parq.iloc[:, :7], how = 'left', on = 'id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:32:15.660872Z","iopub.execute_input":"2024-12-19T09:32:15.661383Z","iopub.status.idle":"2024-12-19T09:34:14.642271Z","shell.execute_reply.started":"2024-12-19T09:32:15.661328Z","shell.execute_reply":"2024-12-19T09:34:14.641085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cont_impute(df, train_check, epoch = 5):\n    if 'id' in df.columns:\n        df = df.drop('id', axis = 1)\n    path = f'{CONT_PATH}/cont_0'\n    file_list = os.listdir(path)\n    file_list = [i for i in file_list if '_0' in i]\n    cont_feat = [i.split('_0')[0] for i in file_list]\n    dicter = {i: [] for i in cont_feat}\n    na_idx = {i: df[df[i].isna()].index for i in cont_feat}\n    df.loc[:, df.select_dtypes(exclude = object).columns] = df[df.select_dtypes(exclude =\\\n                                                                                object).columns].fillna(train_check[df.select_dtypes(exclude = object).columns].median())\n    for i in [p for p in df.columns if 'season' in p]:\n        df[i] = df[i].astype(int)\n    for i in range(epoch + 1):\n        epoch_dir = f'{CONT_PATH}/cont_{i}'\n        for cont in cont_feat:\n            if cont in na_idx.keys():\n                unlabel_vals = df.loc[[i for i in df.index if i in na_idx[cont]]].drop(cont, axis = 1)\n                mdl_path = os.path.join(epoch_dir, f'{cont}_{i}_mdl.cbm')\n                model = CatBoostRegressor()\n                model.load_model(mdl_path)\n                if cont in [i for i in train_check.select_dtypes(exclude = object).columns if any(np.modf(train_check[i].dropna())[0] != 0) is False]:\n                    df.loc[na_idx[cont], cont] = np.clip(np.round(model.predict(unlabel_vals)), train_check[cont].min(), train_check[cont].max())\n                else:\n                    df.loc[na_idx[cont], cont] = model.predict(unlabel_vals).astype('float32')\n                dicter[cont].append(df.loc[na_idx[cont], cont].values)\n    return dicter, na_idx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:34:14.644114Z","iopub.execute_input":"2024-12-19T09:34:14.644606Z","iopub.status.idle":"2024-12-19T09:34:14.659090Z","shell.execute_reply.started":"2024-12-19T09:34:14.644553Z","shell.execute_reply":"2024-12-19T09:34:14.657712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cont_impute_optimal(df, train_check, epoch = 1): \n    checker = [t for t in train_check.copy().select_dtypes(exclude = object).columns if any(np.modf(train_check.copy()[t].dropna())[0] != 0) is False]\n    dict_vals, idxs, = cont_impute(df.copy(), train_check.copy(), epoch) \n    for feat in dict_vals.keys():\n        if feat in checker and len(df[feat].unique()) < 7:\n            df.loc[idxs[feat], feat] = mode(dict_vals[feat])[0]\n        elif feat in checker and len(df[feat].unique() > 7):\n            df.loc[idxs[feat], feat] = np.round(np.mean(dict_vals[feat], axis = 0))\n        else:\n            df.loc[idxs[feat], feat] = np.mean(dict_vals[feat], axis = 0)\n    df.loc[:, df.select_dtypes(exclude = object).columns] = df.select_dtypes(exclude = object).fillna(train_check.select_dtypes(exclude = object).median())\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:34:14.663712Z","iopub.execute_input":"2024-12-19T09:34:14.664350Z","iopub.status.idle":"2024-12-19T09:34:14.682971Z","shell.execute_reply.started":"2024-12-19T09:34:14.664310Z","shell.execute_reply":"2024-12-19T09:34:14.681475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CONT_PATH = '/kaggle/input/features-imputing/new_feat'\nfull_cont = pd.concat([cont_impute_optimal(full_parq.copy().drop('id', axis = 1).iloc[:, :-6], full_parq.copy().drop('id', axis = 1).iloc[:, :-6], 4), full_parq.iloc[:, -6:]], axis = 1)\ntest_cont = pd.concat([cont_impute_optimal(test_parq.copy().drop('id', axis = 1).iloc[:, :-6], full_parq.copy().drop('id', axis = 1).iloc[:, :-6], 4), test_parq.iloc[:, -6:]], axis = 1)\nCONT_PATH = '/kaggle/input/features-imputing/new_feat_cv'\ntrain_cont = pd.concat([cont_impute_optimal(train_parq.copy().drop('id', axis = 1).iloc[:, :-6], train_parq.copy().drop('id', axis = 1).iloc[:, :-6], 9), train_parq.iloc[:, -6:]], axis = 1)\nvalid_cont = pd.concat([cont_impute_optimal(valid_parq.copy().drop('id', axis = 1).iloc[:, :-6], train_parq.copy().drop('id', axis = 1).iloc[:, :-6], 9), valid_parq.iloc[:, -6:]], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:34:14.684556Z","iopub.execute_input":"2024-12-19T09:34:14.684954Z","iopub.status.idle":"2024-12-19T09:35:26.549652Z","shell.execute_reply.started":"2024-12-19T09:34:14.684918Z","shell.execute_reply":"2024-12-19T09:35:26.548309Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Simple_model","metadata":{}},{"cell_type":"code","source":"def final_features(df, trainer = None, clf = False, pca = False):\n    temp = pd.get_dummies(df['basic_demos-enroll_season'], prefix = 'basic_demos-enroll_season').reindex(columns =\n                                                                    [f'basic_demos-enroll_season_{i}' for i in range(4)], fill_value = 0).astype(int)\n    df = pd.concat([df, temp], axis = 1). drop('basic_demos-enroll_season', axis = 1)\n    temp = pd.get_dummies(trainer['basic_demos-enroll_season'], prefix = 'basic_demos-enroll_season').reindex(columns =\n                                                                    [f'basic_demos-enroll_season_{i}' for i in range(4)], fill_value = 0).astype(int)\n    trainer = pd.concat([trainer, temp], axis = 1). drop('basic_demos-enroll_season', axis = 1)\n    if clf is True:\n        return df\n\n    if pca is True:\n        temp = df[[i for i in df.columns if df[i].isna().sum() > 0]]\n        df = df[[i for i in df.columns if i not in temp]]\n        trainer = trainer[[i for i in df.columns if i not in temp]]\n        for col in df.columns:\n            if len(df[col].unique()) > 2:\n                df[col] = (df[col] - trainer[col].min()) / (trainer[col].max() - trainer[col].min())\n                trainer[col] = (trainer[col] - trainer[col].min()) / (trainer[col].max() - trainer[col].min())\n        # mdl = PCA(n_components = 4).fit(trainer)\n        # df = pd.concat([df, pd.DataFrame(mdl.transform(df)), temp], axis = 1)\n        mdl = KMeans(n_clusters = 3, n_init = 10).fit(trainer)\n        df = pd.concat([df, pd.get_dummies(pd.Series(mdl.predict(df)), prefix = 'kmeans').astype(int), temp], axis = 1)\n        \n    df = df[df.iloc[:, :6].columns.to_list() + df.iloc[:, 7: 10].columns.to_list() + ['bia-bia_fat'] + df.iloc[:, 35:].columns.to_list()]\n    return df.drop(['over18', 'physical_sea_diff', 'cgas_sea_diff'], axis = 1)\n\nfull_fin = final_features(full_cont.copy(), full_cont.copy(), pca = False)\ntest_fin = final_features(test_cont.copy(), full_cont.copy(), pca = False)\ntrain_fin = final_features(train_cont.copy(), train_cont.copy(), pca = False)\nvalid_fin = final_features(valid_cont.copy(), train_cont.copy(), pca = False)\n\ntrain_clf = final_features(train_cont.copy(), train_cont.copy(), True)\nvalid_clf = final_features(valid_cont.copy(), train_cont.copy(), True)\nfull_clf = final_features(full_cont.copy(), full_cont.copy(), True)\ntest_clf = final_features(test_cont.copy(), full_cont.copy(), True)\n\n#train_fin = final_features(train_parq.drop('id', axis = 1).copy(), train_parq.drop('id', axis = 1).copy())\n#valid_fin = final_features(valid_parq.drop('id', axis = 1).copy(), train_parq.drop('id', axis = 1).copy())\n#full_fin = final_features(full_parq.drop('id', axis = 1).copy(), full_parq.drop('id', axis = 1).copy())\n#test_fin = final_features(test_parq.drop('id', axis = 1).copy(), full_parq.drop('id', axis = 1).copy())\n\n# train_clf = final_features(train_parq.drop('id', axis = 1).copy(), train_parq.drop('id', axis = 1).copy(), True)\n# valid_clf = final_features(valid_parq.drop('id', axis = 1).copy(), train_parq.drop('id', axis = 1).copy(), True)\n# full_clf = final_features(full_parq.drop('id', axis = 1).copy(), full_parq.drop('id', axis = 1).copy(), True)\n# test_clf = final_features(test_parq.drop('id', axis = 1).copy(), full_parq.drop('id', axis = 1).copy(), True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:02:44.366113Z","iopub.execute_input":"2024-12-19T10:02:44.366558Z","iopub.status.idle":"2024-12-19T10:02:44.451908Z","shell.execute_reply.started":"2024-12-19T10:02:44.366515Z","shell.execute_reply":"2024-12-19T10:02:44.450696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def submit_test(X, y, test_set, model_type = 'regr', agg_process = 'mean'):\n    SPLITS = 5\n    check_metric = 0\n    preds_ = np.zeros((len(test_set), ))\n    total_ = np.zeros((len(X), ))\n    looper_list = [40, 41, 42, 43, 44, 50, 51, 52, 53, 54, 60, 61, 62, 63, 64]\n    \n    for state in tqdm(looper_list):\n        xgb_params = {'n_estimators': 5000, 'tree_method': 'hist', 'eta': 0.05}\n        cat_params = {'thread_count': 4, 'bootstrap_type': 'Bernoulli', 'learning_rate': 0.07}\n        \n        skf = StratifiedKFold(shuffle = True, random_state = state)\n        local_temp = np.zeros((len(X), ))\n        for train_idx, test_idx in skf.split(X, y):\n            X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n            X_test, y_test = X.iloc[test_idx], y.iloc[test_idx]\n            weight = class_weight.compute_class_weight(class_weight = 'balanced', classes = y_train.unique(), y = y_train)\n            weight = {i: weight[idx] for idx, i in enumerate(y_train.unique())}\n            weight = y_train.copy().map(weight).values\n            if model_type == 'class':\n                    model = XGBClassifier(**xgb_params, early_stopping_rounds = 100, objective = 'multi:softmax', eval_metric = xgb_acc, random_state = state).fit(\n                        X_train, y_train, sample_weight = weight, eval_set = [(X_test, y_test)], verbose = 0)\n            else:\n                model = CatBoostRegressor(**cat_params, iterations = 3000, early_stopping_rounds = 100, objective = 'RMSE', eval_metric =  CatKappa(),\n                                    random_state = state).fit(X_train, y_train, sample_weight = weight, eval_set = [(X_test, y_test)], verbose = 0)\n            check_metric += kappa(y_test, model.predict(X_test)) / (5 * len(looper_list))\n            local_temp[test_idx] = model.predict(X_test)\n            if agg_process == 'mean':\n                preds_ += model.predict(test_set) / 50\n            else:\n                preds_ = np.vstack([preds_, model.predict(test_set).reshape(-1, )])\n        total_ = np.vstack([total_, local_temp])\n    print(check_metric)\n    return preds_ #, total_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:02:48.975564Z","iopub.execute_input":"2024-12-19T10:02:48.976028Z","iopub.status.idle":"2024-12-19T10:02:48.989444Z","shell.execute_reply.started":"2024-12-19T10:02:48.975988Z","shell.execute_reply":"2024-12-19T10:02:48.987947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# regr_preds = submit_test(train_fin.copy(), training_['pciat-pciat_total'], valid_fin.copy(), agg_process = 'stack')\n# clf_preds = submit_test(train_clf.copy(), training_.sii, valid_clf.copy(), model_type = 'class', agg_process = 'stack')\n\nregr_preds = submit_test(full_fin.copy(), full_train['pciat-pciat_total'], test_fin.copy(), agg_process = 'stack')\nclf_preds = submit_test(full_clf.copy(), full_train.sii, test_clf.copy(), model_type = 'class', agg_process = 'stack')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:03:12.079962Z","iopub.execute_input":"2024-12-19T10:03:12.080501Z","iopub.status.idle":"2024-12-19T10:03:12.086362Z","shell.execute_reply.started":"2024-12-19T10:03:12.080460Z","shell.execute_reply":"2024-12-19T10:03:12.084926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"regr_ = mode(np.where(regr_preds[1:] < 31, 0, np.where(regr_preds[1:] < 50, 1, np.where(regr_preds[1:] < 80, 2, 3))))\nclf_ = mode(clf_preds[1:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T14:20:57.021233Z","iopub.execute_input":"2024-12-18T14:20:57.022580Z","iopub.status.idle":"2024-12-18T14:20:57.038612Z","shell.execute_reply.started":"2024-12-18T14:20:57.022527Z","shell.execute_reply":"2024-12-18T14:20:57.037398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_ = np.where((clf_[1] > int(len(regr_preds) * 0.95)) & (clf_[0] == 0), clf_[0], \n                  np.where((clf_[1] > int(len(regr_preds) * 0.96)) & (clf_[0] == 3), clf_[0], regr_[0])).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T14:20:59.585577Z","iopub.execute_input":"2024-12-18T14:20:59.585981Z","iopub.status.idle":"2024-12-18T14:20:59.592437Z","shell.execute_reply.started":"2024-12-18T14:20:59.585947Z","shell.execute_reply":"2024-12-18T14:20:59.591268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df = pd.DataFrame({'id': test_df.id, 'sii': preds_})\nsubmit_df.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T14:21:02.964512Z","iopub.execute_input":"2024-12-18T14:21:02.964922Z","iopub.status.idle":"2024-12-18T14:21:02.975636Z","shell.execute_reply.started":"2024-12-18T14:21:02.964887Z","shell.execute_reply":"2024-12-18T14:21:02.974464Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# more models_","metadata":{}},{"cell_type":"code","source":"def tts_custom(train__, y__, valid__, model_type = 'regr'):\n    preds_ = np.zeros((len(valid__), ))\n    for state in tqdm(np.arange(100)):\n        xgb_params = {'n_estimators': 500, 'tree_method': 'hist', 'eta': 0.05}\n        cat_params = {'thread_count': 4, 'bootstrap_type': 'Bernoulli', 'learning_rate': 0.05}\n        # cat_params = {\n        #     'random_state': state, 'eval_metric': CatKappa(), 'iterations': 559, 'learning_rate': 0.051021898437418695, 'min_data_in_leaf': 32.127979806809236, 'depth': 5,\n        #      'l2_leaf_reg': 79.4867300499846, 'grow_policy': 'Lossguide', 'max_leaves': 11, 'border_count': 132, 'bootstrap_type': 'Bayesian'}\n        weight = class_weight.compute_class_weight(class_weight = 'balanced', classes = y__.unique(), y = y__)\n        weight = {i: weight[idx] for idx, i in enumerate(y__.unique())}\n        weight = y__.copy().map(weight).values\n        if model_type == 'class':\n            model = XGBClassifier(**xgb_params, objective = 'multi:softmax', random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        else:\n            model = CatBoostRegressor(**cat_params, iterations = 200, objective = 'RMSE', random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        preds_ = np.vstack([preds_, model.predict(valid__.copy()).reshape(-1, )])\n    return preds_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T10:33:06.886573Z","iopub.execute_input":"2024-12-17T10:33:06.887462Z","iopub.status.idle":"2024-12-17T10:33:06.896428Z","shell.execute_reply.started":"2024-12-17T10:33:06.887424Z","shell.execute_reply":"2024-12-17T10:33:06.894897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# regr_preds = tts_custom(train_fin.copy(), training_['pciat-pciat_total'], valid_fin.copy())\n# clf_preds = tts_custom(train_clf.copy(), training_.sii, valid_clf.copy(), model_type = 'class')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T10:33:07.974690Z","iopub.execute_input":"2024-12-17T10:33:07.975327Z","iopub.status.idle":"2024-12-17T10:43:31.360605Z","shell.execute_reply.started":"2024-12-17T10:33:07.975288Z","shell.execute_reply":"2024-12-17T10:43:31.359722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# regr_ = mode(np.where(regr_preds[1:] < 31, 0, np.where(regr_preds[1:] < 50, 1, np.where(regr_preds[1:] < 80, 2, 3))))\n# clf_ = mode(clf_preds[1:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T10:50:36.913098Z","iopub.execute_input":"2024-12-17T10:50:36.914041Z","iopub.status.idle":"2024-12-17T10:50:36.957043Z","shell.execute_reply.started":"2024-12-17T10:50:36.913997Z","shell.execute_reply":"2024-12-17T10:50:36.956113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# preds_ = np.where((clf_[1] > int(len(regr_preds) * 0.95)) & (clf_[0] == 0) & (regr_[0] < 2), clf_[0], regr_[0]).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T10:51:04.478961Z","iopub.execute_input":"2024-12-17T10:51:04.479377Z","iopub.status.idle":"2024-12-17T10:51:04.485395Z","shell.execute_reply.started":"2024-12-17T10:51:04.479341Z","shell.execute_reply":"2024-12-17T10:51:04.484100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pd.Series(preds_[validing_[validing_.sii == 2].index]).value_counts()\n# pd.Series(clf_[0][validing_[validing_.sii == 3].index]).value_counts()\n# kappa(validing_.sii, preds_)\n\n# pd.Series(clf_[0]).value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:14:31.428720Z","iopub.execute_input":"2024-12-18T12:14:31.429326Z","iopub.status.idle":"2024-12-18T12:14:31.438968Z","shell.execute_reply.started":"2024-12-18T12:14:31.429273Z","shell.execute_reply":"2024-12-18T12:14:31.437647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# optuna","metadata":{}},{"cell_type":"code","source":"def objective(trial, train__, y__, valid__, validy__):\n    preds_ = np.zeros((len(valid__), ))\n    # const_params = {\n    #     'tree_method': 'hist',\n    #     'eval_metric': xgb_kappa,\n    #     'device': 'gpu',\n    #     'objective': 'multi:softmax'}\n    # xgb_params = {\n    #     'n_estimators': trial.suggest_int('n_estimators', 20, 800),\n    #     'eta': trial.suggest_float('eta', 0.01, 0.1),\n    #     'gamma': trial.suggest_float('gamma', 0, 0.4),\n    #     'max_depth': trial.suggest_int('max_depth', 2, 8),\n    #     'alpha': trial.suggest_float('alpha', 0, 100),\n    #     'lambda': trial.suggest_float('lambda', 0, 100),\n    #     'grow_policy': trial.suggest_categorical('grow_policy', ['lossguide', 'depthwise']),\n    #     'max_leaves': trial.suggest_int('max_leaves', 0, 60),\n    #     'feature_selector': trial.suggest_categorical('feature_selector', ['cyclic', 'shuffle', 'random', 'greedy', 'thrifty']),\n    #     'top_k': trial.suggest_int('top_k', 1, 30)\n    # }\n\n    const_params = {\n        'eval_metric': CatKappa(),\n        'objective': 'RMSE'}\n    cat_params = {\n        'iterations': trial.suggest_int('iterations', 20, 900),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.12),\n        'min_data_in_leaf': trial.suggest_float('min_data_in_leaf', 0, 50),\n        'depth': trial.suggest_int('depth', 2, 7),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 0, 100),\n        'grow_policy': trial.suggest_categorical('grow_policy', ['Lossguide', 'Depthwise', 'SymmetricTree']),\n        'max_leaves': trial.suggest_int('max_leaves', 0, 60),\n        'border_count': trial.suggest_int('border_count', 30, 400),\n        'bootstrap_type': trial.suggest_categorical('bootstrap_type', ['Bayesian', 'Bernoulli'])}\n    # if xgb_params['feature_selector'] not in  ['greedy', 'thrifty']:\n    #     xgb_params = {i: j for i, j in xgb_params.items() if i not in ['feature_selector', 'top_k']}\n    if cat_params['grow_policy'] in ['Depthwise', 'SymmetricTree']:\n        cat_params = {i: j for i, j in cat_params.items() if i not in ['max_leaves']}\n        \n    for state in tqdm(np.arange(100)):\n        weight = class_weight.compute_class_weight(class_weight = 'balanced', classes = y__.unique(), y = y__)\n        weight = {i: weight[idx] for idx, i in enumerate(y__.unique())}\n        weight = y__.copy().map(weight).values\n        model = CatBoostRegressor(**const_params, **cat_params, random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        # model = CatBoostClassifier(**const_params, **cat_params, random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        # model = XGBClassifier(**xgb_params, **const_params, random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        # model = XGBRegressor(**xgb_params, **const_params, random_state = state).fit(train__, y__, sample_weight = weight, verbose = 0)\n        preds_ = np.vstack([preds_, model.predict(valid__).reshape(-1, )])\n    temp_preds = mode(np.where(preds_[1:] < 31, 0, np.where(preds_[1:] < 50, 1, np.where(preds_[1:] < 80, 2, 3))))[0]\n    # temp_preds = mode(preds_[1:])[0]\n    \n    kap_score = kappa(validy__, temp_preds)\n    param_list.append((kap_score, cat_params, pd.Series(temp_preds).value_counts().to_dict()))\n    acc_list.append((accuracy_score(validy__, temp_preds), cat_params, pd.Series(temp_preds).value_counts().to_dict()))\n    if len(param_list) > 10:\n        param_list.pop(param_list.index(min(param_list, key = lambda x: x[0])))\n        acc_list.pop(acc_list.index(min(acc_list, key = lambda x: x[0])))\n    return kap_score\ndef optuna_tts():\n    global param_list, acc_list\n    param_list = []\n    acc_list = []\n    study = optuna.create_study(direction = 'maximize')\n    # obj = partial(objective, train__ = kmeans_train.copy(), y__ = training_['pciat-pciat_total'], valid__ = kmeans_valid.copy(), validy__ = validing_['pciat-pciat_total'])\n    obj = partial(objective, train__ = test_train.copy(), y__ = training_['pciat-pciat_total'], valid__ = test_valid.copy(), validy__ = validing_['pciat-pciat_total'])\n    # obj = partial(objective, train__ = kmeans_train.copy(), y__ = training_.sii, valid__ = kmeans_valid.copy(), validy__ = validing_.sii)\n    study.optimize(obj, n_trials = 65)\n    \n    print(*param_list, sep = '\\n')\n    print('\\n', *acc_list, sep = '\\n')\n# optuna_tts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T23:56:26.647786Z","iopub.execute_input":"2024-12-15T23:56:26.648630Z","iopub.status.idle":"2024-12-15T23:56:26.664045Z","shell.execute_reply.started":"2024-12-15T23:56:26.648591Z","shell.execute_reply":"2024-12-15T23:56:26.662933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial, train__, y__, clf_preds):\n    const_params = {\n        'eval_metric': CatKappa(),\n        'objective': 'RMSE',\n        'iterations': 1000,\n        'bootstrap_type': 'Bernoulli'}\n    cat_params = {\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.2)}\n\n    kap_score = 0\n    looper_list = [40, 41, 42, 43, 44, 50, 51, 52, 53, 54, 60, 61, 62, 63, 64]\n    preds1 = np.zeros((len(train__), ))\n    for state in tqdm(looper_list):\n        skf = StratifiedKFold(shuffle = True, random_state = state)\n        local_preds = np.zeros((len(train__), ))\n        for train_idx, test_idx in skf.split(train__, y__):\n            X_train, y_train = train__.iloc[train_idx], y__.iloc[train_idx]\n            X_test, y_test = train__.iloc[test_idx], y__.iloc[test_idx]\n            weight = class_weight.compute_class_weight(class_weight = 'balanced', classes = y_train.unique(), y = y_train)\n            weight = {i: weight[idx] for idx, i in enumerate(y_train.unique())}\n            weight = y_train.copy().map(weight).values\n            model = CatBoostRegressor(**const_params, **cat_params, random_state = state, early_stopping_rounds=100).fit(X_train, y_train, \n                                                                                                        eval_set = [(X_test, y_test)], sample_weight = weight, verbose = 0)\n            local_preds[test_idx] = model.predict(X_test)\n        preds1 = np.vstack([preds1, local_preds])\n    \n\n    regrss_ = mode(np.where(preds1[1:] < 31, 0, np.where(preds1[1:] < 50, 1, np.where(preds1[1:] < 80, 2, 3))))\n    predsss = np.where((clf_preds[1] > int(13)) & (clf_preds[0] == 0), clf_preds[0], regrss_[0]).astype(int)    \n    kap_score =  kappa(y__, regrss_[0])\n    \n    param_list.append((kap_score, cat_params))\n    if len(param_list) > 10:\n        param_list.pop(param_list.index(min(param_list, key = lambda x: x[0])))\n    return kap_score\ndef optuna_cv():\n    global param_list\n    param_list = []\n    study = optuna.create_study(direction = 'maximize')\n    # obj = partial(objective, train__ = full_fin.copy(), y__ = full_train['pciat-pciat_total'])\n    obj = partial(objective, train__ = full_fin.copy(), y__ = full_train['pciat-pciat_total'], clf_preds = clftest)\n    study.optimize(obj, n_trials = 150)\n    print(*param_list, sep = '\\n')\n# optuna_cv()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:51:31.197409Z","iopub.execute_input":"2024-12-18T13:51:31.197847Z","iopub.status.idle":"2024-12-18T13:51:31.212998Z","shell.execute_reply.started":"2024-12-18T13:51:31.197810Z","shell.execute_reply":"2024-12-18T13:51:31.211643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"{'learning_rate': 0.41583212399996156} - best 0.488...\n'learning_rate': 0.4230343258109492 -2nd best 0.486...\n\n 0.47500504835199353 and parameters: {'learning_rate': 0.10734701441884839}\n 0.4772620940458505 and parameters: {'learning_rate': 0.09752897972948855}","metadata":{}},{"cell_type":"code","source":"def iterator_do(train__, y__):\n    score = 0\n    cat_params = {'thread_count': 4, 'bootstrap_type': 'Bernoulli', 'iterations': 3000, 'learning_rate': 0.07}\n    for state in tqdm([40, 41, 42, 43, 44]):\n        skf = StratifiedKFold(shuffle = True, random_state = state)\n        for train_idx, test_idx in skf.split(train__, y__):\n            X_train, y_train = train__.iloc[train_idx], y__.iloc[train_idx]\n            X_test, y_test = train__.iloc[test_idx], y__.iloc[test_idx]\n            weight = class_weight.compute_class_weight(class_weight = 'balanced', classes = y_train.unique(), y = y_train)\n            weight = {i: weight[idx] for idx, i in enumerate(y_train.unique())}\n            weight = y_train.copy().map(weight).values\n            model = CatBoostRegressor(**cat_params, early_stopping_rounds = 100, objective = 'RMSE', eval_metric = CatKappa(), random_state = state).fit(X_train, y_train,\n                                                                                                    sample_weight = weight, eval_set = [(X_test, y_test)], verbose = 0)\n            score += kappa(y_test, model.predict(X_test)) / 25\n    return score\n\ndef iterator(train__, y__):\n    kap_score = iterator_do(train__.copy(), y__.copy())\n    print(f'start_score: {kap_score}')\n    all_feats = []\n    while True:\n        best_score = kap_score\n        curr_feat = \"\"\n        for col in train__.columns:\n            local_score = iterator_do(train__.copy().drop(all_feats + [col], axis = 1), y__.copy())\n            if local_score > best_score:\n                best_score = local_score\n                curr_feat = col\n            print(f'{col}: --{local_score}')\n        if len(curr_feat) == 0:\n            print(f'========final_score: {kap_score} and feat_list: {all_feats}========')\n            break\n        all_feats.append(curr_feat)\n        kap_score = best_score\n        print(f'============{kap_score}, {all_feats}============')\n            \n# iterator(kmeans_full.copy(), full_train['pciat-pciat_total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T15:41:11.070379Z","iopub.execute_input":"2024-12-14T15:41:11.070786Z","iopub.status.idle":"2024-12-14T20:15:01.040330Z","shell.execute_reply.started":"2024-12-14T15:41:11.070751Z","shell.execute_reply":"2024-12-14T20:15:01.038765Z"}},"outputs":[],"execution_count":null}]}