{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":81933,"databundleVersionId":9643020},{"sourceType":"datasetVersion","sourceId":10242526,"datasetId":6267160,"databundleVersionId":10538517}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Child Mind Institute - Problematic Internet Use","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/cmu-piu-dataset/packages/lightgbm-4.5.0-py3-none-manylinux_2_28_x86_64.whl --no-index --find-links /kaggle/input/cmu-piu-dataset/packages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:32:56.682519Z","iopub.execute_input":"2024-12-19T06:32:56.682992Z","iopub.status.idle":"2024-12-19T06:33:09.883149Z","shell.execute_reply.started":"2024-12-19T06:32:56.682934Z","shell.execute_reply":"2024-12-19T06:33:09.881682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import yaml\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport json\nimport numpy as np\nimport pandas as pd\nfrom scipy.optimize import minimize\nfrom scipy.stats import mode\nimport lightgbm as lgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:33:09.886022Z","iopub.execute_input":"2024-12-19T06:33:09.886554Z","iopub.status.idle":"2024-12-19T06:33:12.549350Z","shell.execute_reply.started":"2024-12-19T06:33:09.886491Z","shell.execute_reply":"2024-12-19T06:33:12.548130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_rows', 200)\npd.set_option('display.max_columns', 200)\npd.set_option('display.width', 200)\n\ncompetition_dataset_directory = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\nexternal_dataset_directory = Path('/kaggle/input/cmu-piu-dataset')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:33:12.551378Z","iopub.execute_input":"2024-12-19T06:33:12.552206Z","iopub.status.idle":"2024-12-19T06:33:12.558329Z","shell.execute_reply.started":"2024-12-19T06:33:12.552150Z","shell.execute_reply":"2024-12-19T06:33:12.557040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(competition_dataset_directory / 'test.csv')\nprint(f'Dataset Shape: {df.shape}')\ndisplay(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:51:24.292531Z","iopub.execute_input":"2024-12-19T06:51:24.292988Z","iopub.status.idle":"2024-12-19T06:51:24.383828Z","shell.execute_reply.started":"2024-12-19T06:51:24.292951Z","shell.execute_reply":"2024-12-19T06:51:24.382659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Preprocessing","metadata":{}},{"cell_type":"code","source":"def clean(df):\n\n    df.loc[df['Physical-Weight'] == 0, 'Physical-Weight'] = np.nan\n    df.loc[df['Physical-BMI'] == 0, 'Physical-BMI'] = np.nan\n\n    bmi_missing_mask1 = df['Physical-BMI'].notna() & df['BIA-BIA_BMI'].isnull()\n    df.loc[bmi_missing_mask1, 'BIA-BIA_BMI'] = df.loc[bmi_missing_mask1, 'Physical-BMI']\n    bmi_missing_mask2 = df['Physical-BMI'].isnull() & df['BIA-BIA_BMI'].notna()\n    df.loc[bmi_missing_mask2, 'Physical-BMI'] = df.loc[bmi_missing_mask2, 'BIA-BIA_BMI']\n    df['bmi'] = df['Physical-BMI'] * 0.5 + df['BIA-BIA_BMI'] * 0.5\n\n    df['Fitness_Endurance-Time_Sec'] = df['Fitness_Endurance-Time_Sec'] + df['Fitness_Endurance-Time_Mins'] * 60\n\n    return df\n\ndf = clean(df=df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:51:25.556839Z","iopub.execute_input":"2024-12-19T06:51:25.557218Z","iopub.status.idle":"2024-12-19T06:51:25.573879Z","shell.execute_reply.started":"2024-12-19T06:51:25.557185Z","shell.execute_reply":"2024-12-19T06:51:25.571896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Models","metadata":{}},{"cell_type":"code","source":"def load_model(model_directory):\n\n    models = {}\n\n    for model_path in tqdm(sorted(list(model_directory.glob('model*')))):\n        model_path = str(model_path)\n        if 'lightgbm' in model_path:\n            model = lgb.Booster(model_file=model_path)\n        else:\n            raise ValueError('Invalid model type')\n        \n        model_file_name = model_path.split('/')[-1].split('.')[0]\n        seed = int(model_file_name.split('_')[-1])\n        models[model_file_name] = model\n\n    config = yaml.load(open(model_directory / 'config.yaml'), Loader=yaml.FullLoader)\n    with open(model_directory / 'thresholds.json', mode='r') as f:\n        thresholds = json.load(f)\n\n    print(f'Loaded models, config and thresholds from {model_directory}')\n    \n    return config, models, thresholds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:36:44.371619Z","iopub.execute_input":"2024-12-19T06:36:44.372623Z","iopub.status.idle":"2024-12-19T06:36:44.379878Z","shell.execute_reply.started":"2024-12-19T06:36:44.372539Z","shell.execute_reply":"2024-12-19T06:36:44.378654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_r_h_ptn_1_model_path = external_dataset_directory / 'lightgbm_regression_huber_pciat_total_normalized_1'\nlgb_r_h_ptn_1_config, lgb_r_h_ptn_1_models, lgb_r_h_ptn_1_thresholds = load_model(lgb_r_h_ptn_1_model_path)\nlgb_r_h_ptn_1_models = {model_file_name: model for model_file_name, model in lgb_r_h_ptn_1_models.items() if 'fold' in model_file_name}\n\nlgb_r_h_ptn_2_model_path = external_dataset_directory / 'lightgbm_regression_huber_pciat_total_normalized_2'\nlgb_r_h_ptn_2_config, lgb_r_h_ptn_2_models, lgb_r_h_ptn_2_thresholds = load_model(lgb_r_h_ptn_2_model_path)\nlgb_r_h_ptn_2_models = {model_file_name: model for model_file_name, model in lgb_r_h_ptn_2_models.items() if 'fold' in model_file_name}\n\nlgb_r_h_ptn_3_model_path = external_dataset_directory / 'lightgbm_regression_huber_pciat_total_normalized_3'\nlgb_r_h_ptn_3_config, lgb_r_h_ptn_3_models, lgb_r_h_ptn_3_thresholds = load_model(lgb_r_h_ptn_3_model_path)\nlgb_r_h_ptn_3_models = {model_file_name: model for model_file_name, model in lgb_r_h_ptn_3_models.items() if 'fold' in model_file_name}\n\nlgb_r_mse_ptn_1_model_path = external_dataset_directory / 'lightgbm_regression_mse_pciat_total_normalized_1'\nlgb_r_mse_ptn_1_config, lgb_r_mse_ptn_1_models, lgb_r_mse_ptn_1_thresholds = load_model(lgb_r_mse_ptn_1_model_path)\nlgb_r_mse_ptn_1_models = {model_file_name: model for model_file_name, model in lgb_r_mse_ptn_1_models.items() if 'fold' in model_file_name}\n\nlgb_r_mse_ptn_2_model_path = external_dataset_directory / 'lightgbm_regression_mse_pciat_total_normalized_2'\nlgb_r_mse_ptn_2_config, lgb_r_mse_ptn_2_models, lgb_r_mse_ptn_2_thresholds = load_model(lgb_r_mse_ptn_2_model_path)\nlgb_r_mse_ptn_2_models = {model_file_name: model for model_file_name, model in lgb_r_mse_ptn_2_models.items() if 'fold' in model_file_name}\n\nlgb_r_mse_ptn_3_model_path = external_dataset_directory / 'lightgbm_regression_mse_pciat_total_normalized_3'\nlgb_r_mse_ptn_3_config, lgb_r_mse_ptn_3_models, lgb_r_mse_ptn_3_thresholds = load_model(lgb_r_mse_ptn_3_model_path)\nlgb_r_mse_ptn_3_models = {model_file_name: model for model_file_name, model in lgb_r_mse_ptn_3_models.items() if 'fold' in model_file_name}\n\nlgb_c_ce_ptn_1_model_path = external_dataset_directory / 'lightgbm_classification_ce_pciat_total_normalized_1'\nlgb_c_ce_ptn_1_config, lgb_c_ce_ptn_1_models, lgb_c_ce_ptn_1_thresholds = load_model(lgb_c_ce_ptn_1_model_path)\nlgb_c_ce_ptn_1_models = {model_file_name: model for model_file_name, model in lgb_c_ce_ptn_1_models.items() if 'fold' in model_file_name}\n\nlgb_c_ce_ptn_2_model_path = external_dataset_directory / 'lightgbm_classification_ce_pciat_total_normalized_2'\nlgb_c_ce_ptn_2_config, lgb_c_ce_ptn_2_models, lgb_c_ce_ptn_2_thresholds = load_model(lgb_c_ce_ptn_2_model_path)\nlgb_c_ce_ptn_2_models = {model_file_name: model for model_file_name, model in lgb_c_ce_ptn_2_models.items() if 'fold' in model_file_name}\n\nlgb_c_ce_ptn_3_model_path = external_dataset_directory / 'lightgbm_classification_ce_pciat_total_normalized_3'\nlgb_c_ce_ptn_3_config, lgb_c_ce_ptn_3_models, lgb_c_ce_ptn_3_thresholds = load_model(lgb_c_ce_ptn_3_model_path)\nlgb_c_ce_ptn_3_models = {model_file_name: model for model_file_name, model in lgb_c_ce_ptn_3_models.items() if 'fold' in model_file_name}\n\nlgb_c_cel_ptn_1_model_path = external_dataset_directory / 'lightgbm_classification_cel_pciat_total_normalized_1'\nlgb_c_cel_ptn_1_config, lgb_c_cel_ptn_1_models, lgb_c_cel_ptn_1_thresholds = load_model(lgb_c_cel_ptn_1_model_path)\nlgb_c_cel_ptn_1_models = {model_file_name: model for model_file_name, model in lgb_c_cel_ptn_1_models.items() if 'fold' in model_file_name}\n\nlgb_c_cel_ptn_2_model_path = external_dataset_directory / 'lightgbm_classification_cel_pciat_total_normalized_2'\nlgb_c_cel_ptn_2_config, lgb_c_cel_ptn_2_models, lgb_c_cel_ptn_2_thresholds = load_model(lgb_c_cel_ptn_2_model_path)\nlgb_c_cel_ptn_2_models = {model_file_name: model for model_file_name, model in lgb_c_cel_ptn_2_models.items() if 'fold' in model_file_name}\n\nlgb_c_cel_ptn_3_model_path = external_dataset_directory / 'lightgbm_classification_cel_pciat_total_normalized_3'\nlgb_c_cel_ptn_3_config, lgb_c_cel_ptn_3_models, lgb_c_cel_ptn_3_thresholds = load_model(lgb_c_cel_ptn_3_model_path)\nlgb_c_cel_ptn_3_models = {model_file_name: model for model_file_name, model in lgb_c_cel_ptn_3_models.items() if 'fold' in model_file_name}\n\nlightgbm_model_names = [\n    'lgb_r_h_ptn_1',\n    'lgb_r_h_ptn_2',\n    'lgb_r_h_ptn_3',\n    'lgb_r_mse_ptn_1',\n    'lgb_r_mse_ptn_2',\n    'lgb_r_mse_ptn_3',\n    'lgb_c_ce_ptn_1',\n    'lgb_c_ce_ptn_2',\n    'lgb_c_ce_ptn_3',\n    'lgb_c_cel_ptn_1',\n    'lgb_c_cel_ptn_2',\n    'lgb_c_cel_ptn_3'\n]\nlightgbm_configs = [\n    lgb_r_h_ptn_1_config,\n    lgb_r_h_ptn_2_config,\n    lgb_r_h_ptn_3_config,\n    lgb_r_mse_ptn_1_config,\n    lgb_r_mse_ptn_2_config,\n    lgb_r_mse_ptn_3_config,\n    lgb_c_ce_ptn_1_config,\n    lgb_c_ce_ptn_2_config,\n    lgb_c_ce_ptn_3_config,\n    lgb_c_cel_ptn_1_config,\n    lgb_c_cel_ptn_2_config,\n    lgb_c_cel_ptn_3_config\n]\nlightgbm_models = [\n    lgb_r_h_ptn_1_models,\n    lgb_r_h_ptn_2_models,\n    lgb_r_h_ptn_3_models,\n    lgb_r_mse_ptn_1_models,\n    lgb_r_mse_ptn_2_models,\n    lgb_r_mse_ptn_3_models,\n    lgb_c_ce_ptn_1_models,\n    lgb_c_ce_ptn_2_models,\n    lgb_c_ce_ptn_3_models,\n    lgb_c_cel_ptn_1_models,\n    lgb_c_cel_ptn_2_models,\n    lgb_c_cel_ptn_3_models\n]\nlightgbm_thresholds = [\n    lgb_r_h_ptn_1_thresholds,\n    lgb_r_h_ptn_2_thresholds,\n    lgb_r_h_ptn_3_thresholds,\n    lgb_r_mse_ptn_1_thresholds,\n    lgb_r_mse_ptn_2_thresholds,\n    lgb_r_mse_ptn_3_thresholds,\n    lgb_c_ce_ptn_1_thresholds,\n    lgb_c_ce_ptn_2_thresholds,\n    lgb_c_ce_ptn_3_thresholds,\n    lgb_c_cel_ptn_1_thresholds,\n    lgb_c_cel_ptn_2_thresholds,\n    lgb_c_cel_ptn_3_thresholds\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:36:45.155053Z","iopub.execute_input":"2024-12-19T06:36:45.155421Z","iopub.status.idle":"2024-12-19T06:36:57.642904Z","shell.execute_reply.started":"2024-12-19T06:36:45.155388Z","shell.execute_reply":"2024-12-19T06:36:57.641547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Inference","metadata":{}},{"cell_type":"code","source":"def round_predictions(y_pred, thresholds):\n\n    y_pred = np.where(\n        y_pred < thresholds[0],\n        0,\n        np.where(\n            y_pred < thresholds[1],\n            1,\n            np.where(\n                y_pred < thresholds[2],\n                2,\n                3\n            )\n        )\n    )\n\n    return y_pred\n\n\ndef lightgbm_predict(df, model_name, config, models, thresholds):\n\n    prediction_column = f'{model_name}_prediction'\n    rounded_prediction_column = f'{model_name}_prediction_rounded'\n    \n    df[prediction_column] = 0.\n\n    for model in list(models.values()):\n        model_predictions = model.predict(\n            df.loc[:, config['training']['features']],\n            num_iteration=config['fit_parameters']['boosting_rounds']\n        )\n        df.loc[:, prediction_column] += model_predictions / len(models)\n\n    df[rounded_prediction_column] = round_predictions(\n        df[prediction_column],\n        thresholds=thresholds\n    )\n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:51:28.638707Z","iopub.execute_input":"2024-12-19T06:51:28.639091Z","iopub.status.idle":"2024-12-19T06:51:28.648040Z","shell.execute_reply.started":"2024-12-19T06:51:28.639059Z","shell.execute_reply":"2024-12-19T06:51:28.646547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for model_name, config, models, thresholds in tqdm(zip(lightgbm_model_names, lightgbm_configs, lightgbm_models, lightgbm_thresholds), total=len(lightgbm_model_names)):\n    print(f'Predicting with {model_name} (Thresholds: {thresholds})')\n    df = lightgbm_predict(\n        df=df,\n        model_name=model_name,\n        config=config,\n        models=models,\n        thresholds=thresholds\n        \n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T06:51:29.901518Z","iopub.execute_input":"2024-12-19T06:51:29.902552Z","iopub.status.idle":"2024-12-19T06:51:31.356033Z","shell.execute_reply.started":"2024-12-19T06:51:29.902497Z","shell.execute_reply":"2024-12-19T06:51:31.354776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"soft_prediction_columns = [\n    column for column in df.columns\n    if 'prediction' in column and 'rounded' not in column\n][:]\n\nclass_prediction_columns = [\n    column for column in df.columns\n    if 'prediction' in column and 'rounded' in column\n][:]\n\ndisplay(df[soft_prediction_columns])\ndisplay(df[class_prediction_columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:04:59.984085Z","iopub.execute_input":"2024-12-19T08:04:59.984596Z","iopub.status.idle":"2024-12-19T08:05:00.031425Z","shell.execute_reply.started":"2024-12-19T08:04:59.984519Z","shell.execute_reply":"2024-12-19T08:05:00.030168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Ensemble","metadata":{}},{"cell_type":"code","source":"df['voting_prediction'] = mode(df[class_prediction_columns], axis=1)[0]\ndisplay(df['voting_prediction'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:05:04.799719Z","iopub.execute_input":"2024-12-19T08:05:04.800118Z","iopub.status.idle":"2024-12-19T08:05:04.817807Z","shell.execute_reply.started":"2024-12-19T08:05:04.800085Z","shell.execute_reply":"2024-12-19T08:05:04.816408Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Submission","metadata":{}},{"cell_type":"code","source":"df['submission_prediction'] = df['voting_prediction'].values\n\ndisplay(df['submission_prediction'])\ndisplay(df['submission_prediction'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:05:07.323598Z","iopub.execute_input":"2024-12-19T08:05:07.324011Z","iopub.status.idle":"2024-12-19T08:05:07.339122Z","shell.execute_reply.started":"2024-12-19T08:05:07.323974Z","shell.execute_reply":"2024-12-19T08:05:07.337730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_submission = df.loc[:, ['id', 'submission_prediction']].rename(columns={'submission_prediction': 'sii'})\ndf_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T08:05:07.562064Z","iopub.execute_input":"2024-12-19T08:05:07.562452Z","iopub.status.idle":"2024-12-19T08:05:07.576409Z","shell.execute_reply.started":"2024-12-19T08:05:07.562419Z","shell.execute_reply":"2024-12-19T08:05:07.575131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T07:50:38.397099Z","iopub.execute_input":"2024-12-19T07:50:38.397585Z","iopub.status.idle":"2024-12-19T07:50:38.407632Z","shell.execute_reply.started":"2024-12-19T07:50:38.397516Z","shell.execute_reply":"2024-12-19T07:50:38.405516Z"}},"outputs":[],"execution_count":null}]}