{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ANALYSIS | CAT + LGBM + XGB | OPTUNE\n\nThere is Japanese here and there","metadata":{}},{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport matplotlib.pyplot as plt\nimport missingno as msno\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport xgboost as xgb\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-24T12:44:38.360647Z","iopub.execute_input":"2024-09-24T12:44:38.361238Z","iopub.status.idle":"2024-09-24T12:44:44.675662Z","shell.execute_reply.started":"2024-09-24T12:44:38.361178Z","shell.execute_reply":"2024-09-24T12:44:44.674095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday','sii']\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n 'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c : \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n        \n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n    \nfor col in cat_c:\n    all_values = pd.concat([train[col], test[col]]).unique()\n    mapping = {value: idx for idx, value in enumerate(all_values)}\n\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:44.678188Z","iopub.execute_input":"2024-09-24T12:44:44.679018Z","iopub.status.idle":"2024-09-24T12:44:44.921501Z","shell.execute_reply.started":"2024-09-24T12:44:44.678958Z","shell.execute_reply":"2024-09-24T12:44:44.919834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ANALYSIS","metadata":{}},{"cell_type":"markdown","source":"## histgrams","metadata":{}},{"cell_type":"code","source":"%%time\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:44.923027Z","iopub.execute_input":"2024-09-24T12:44:44.923515Z","iopub.status.idle":"2024-09-24T12:44:45.037210Z","shell.execute_reply.started":"2024-09-24T12:44:44.923466Z","shell.execute_reply":"2024-09-24T12:44:45.035673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# ヒストグラムを作成する\ntrain.hist(figsize=(15, 10), bins=20, xlabelsize=8, ylabelsize=8)\n\n# グラフのレイアウトを調整して重ならないようにする\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:45.041163Z","iopub.execute_input":"2024-09-24T12:44:45.041817Z","iopub.status.idle":"2024-09-24T12:44:58.846761Z","shell.execute_reply.started":"2024-09-24T12:44:45.041747Z","shell.execute_reply":"2024-09-24T12:44:58.845290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sii\"].hist()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:58.848644Z","iopub.execute_input":"2024-09-24T12:44:58.849131Z","iopub.status.idle":"2024-09-24T12:44:59.218580Z","shell.execute_reply.started":"2024-09-24T12:44:58.849084Z","shell.execute_reply":"2024-09-24T12:44:59.217012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## missing values","metadata":{}},{"cell_type":"code","source":"missing_percent_train = train.isnull().mean() * 100\nmissing_percent_train","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:59.220820Z","iopub.execute_input":"2024-09-24T12:44:59.221472Z","iopub.status.idle":"2024-09-24T12:44:59.240108Z","shell.execute_reply.started":"2024-09-24T12:44:59.221402Z","shell.execute_reply":"2024-09-24T12:44:59.238625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percent_test = test.isnull().mean() * 100\nmissing_percent_test","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:59.242040Z","iopub.execute_input":"2024-09-24T12:44:59.242587Z","iopub.status.idle":"2024-09-24T12:44:59.268318Z","shell.execute_reply.started":"2024-09-24T12:44:59.242523Z","shell.execute_reply":"2024-09-24T12:44:59.266783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:44:59.270472Z","iopub.execute_input":"2024-09-24T12:44:59.271112Z","iopub.status.idle":"2024-09-24T12:45:00.023007Z","shell.execute_reply.started":"2024-09-24T12:44:59.271042Z","shell.execute_reply":"2024-09-24T12:45:00.021121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(test)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:00.025355Z","iopub.execute_input":"2024-09-24T12:45:00.025992Z","iopub.status.idle":"2024-09-24T12:45:00.682598Z","shell.execute_reply.started":"2024-09-24T12:45:00.025919Z","shell.execute_reply":"2024-09-24T12:45:00.680725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## interaction feature","metadata":{}},{"cell_type":"code","source":"def create_interaction_features(df, feature_pairs):\n    for feature1, feature2 in feature_pairs:\n        new_feature_name = f\"{feature1}_x_{feature2}\"\n        df[new_feature_name] = df[feature1] * df[feature2]\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:00.687712Z","iopub.execute_input":"2024-09-24T12:45:00.688313Z","iopub.status.idle":"2024-09-24T12:45:00.696585Z","shell.execute_reply.started":"2024-09-24T12:45:00.688244Z","shell.execute_reply":"2024-09-24T12:45:00.694920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_pairs = [\n    ('PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age'),\n    ('Basic_Demos-Age', 'SDS-SDS_Total_T'),\n    ('FGC-FGC_SRR_Zone', 'SDS-SDS_Total_T'),\n    ('BIA-BIA_BMC', 'Physical-HeartRate'),\n    ('Fitness_Endurance-Season', 'Physical-Waist_Circumference'),\n    ('BIA-BIA_Fat', 'Physical-BMI'),\n    ('PreInt_EduHx-Season', 'Fitness_Endurance-Season'),\n    ('SDS-SDS_Total_T', 'Physical-Systolic_BP'),\n    ('Basic_Demos-Sex', 'FGC-FGC_PU_Zone')\n]\n\n\ntrain = create_interaction_features(train, feature_pairs)\ntest = create_interaction_features(test, feature_pairs)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:00.698400Z","iopub.execute_input":"2024-09-24T12:45:00.698940Z","iopub.status.idle":"2024-09-24T12:45:00.730338Z","shell.execute_reply.started":"2024-09-24T12:45:00.698892Z","shell.execute_reply":"2024-09-24T12:45:00.728178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"X = train.drop(['sii'], axis=1)\ny = train['sii']\n# モデルを学習した後のコード\nXGBoost = xgb.XGBRegressor(random_state=SEED)\nXGBoost.fit(X, y)\n\n# 特徴量重要度を取得\nimportance = XGBoost.feature_importances_\n\n# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n# 特徴量重要度を降順に並び替え\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n# プロット\nplt.figure(figsize=(10, 20))\nplt.barh(importance_df['Feature'], importance_df['Importance'])\nplt.xlabel('Feature Importance')\nplt.ylabel('Features')\nplt.title('Feature Importance in XGBoost')\nplt.gca().invert_yaxis()  # 重要度が高いものを上に\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:00.732597Z","iopub.execute_input":"2024-09-24T12:45:00.733246Z","iopub.status.idle":"2024-09-24T12:45:02.980706Z","shell.execute_reply.started":"2024-09-24T12:45:00.733177Z","shell.execute_reply":"2024-09-24T12:45:02.979194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This plot shows the feature importance for an XGBoost model, ranked from the most important feature to the least important. Key insights from this visualization include:\n\nTop Features:\n\nThe most important feature is PreInt_EduHx-computerinternet_hoursday, indicating that the time spent on computer/internet usage per day plays the largest role in the model's predictions.\nOther highly ranked features include Basic_Demos-Age, FGC-FGC_SRR_Zone, and SDS-SDS_Total_T, which suggest that age and specific health or fitness-related scores are also crucial for the model's decisions.\n\nFeature Distribution:\n\nThere is a wide range of feature importances, but most of them contribute relatively equally beyond the top few features.\nThe long tail of features implies that while the top few features dominate, a large number of other features still contribute in small amounts to the predictions.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理（特徴量重要度と欠損値の割合を結合）\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\nmissing_df = pd.DataFrame({'Feature': missing_percent_train.index, 'MissingPercent': missing_percent_train.values})\ncombined_df = pd.merge(importance_df, missing_df, on='Feature')\n\n# 散布図を作成\nplt.figure(figsize=(10, 6))\nplt.scatter(combined_df['MissingPercent'], combined_df['Importance'], alpha=0.7)\nplt.xlabel('Missing Percentage (%)')\nplt.ylabel('Feature Importance')\nplt.title('Feature Importance vs Missing Percentage')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:02.982635Z","iopub.execute_input":"2024-09-24T12:45:02.983172Z","iopub.status.idle":"2024-09-24T12:45:03.381190Z","shell.execute_reply.started":"2024-09-24T12:45:02.983117Z","shell.execute_reply":"2024-09-24T12:45:03.379201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission, tKappa","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:03.383239Z","iopub.execute_input":"2024-09-24T12:45:03.383991Z","iopub.status.idle":"2024-09-24T12:45:03.407280Z","shell.execute_reply.started":"2024-09-24T12:45:03.383906Z","shell.execute_reply":"2024-09-24T12:45:03.405567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params_lgbm = {'num_leaves': 60, 'learning_rate': 0.011747572224219955, 'n_estimators': 993, 'max_depth': 1, 'min_child_weight': 0.0025036281384857462, 'subsample': 0.8252622287203014, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.7153672744430527, 'reg_lambda': 0.12158717311465662}\nbest_params_xgb = {'learning_rate': 0.007356059931165658, 'max_depth': 3, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\nbest_params_catboost = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'depth': 6, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:03.409189Z","iopub.execute_input":"2024-09-24T12:45:03.409758Z","iopub.status.idle":"2024-09-24T12:45:03.425565Z","shell.execute_reply.started":"2024-09-24T12:45:03.409706Z","shell.execute_reply":"2024-09-24T12:45:03.423949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# LightGBM\nLight = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\nSubmission_LGBM, k_lgbm = TrainML(Light, test)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:03.427411Z","iopub.execute_input":"2024-09-24T12:45:03.427953Z","iopub.status.idle":"2024-09-24T12:45:07.883600Z","shell.execute_reply.started":"2024-09-24T12:45:03.427887Z","shell.execute_reply":"2024-09-24T12:45:07.881989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XGBoost\nXGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED)\nSubmission_XGB, k_xgb = TrainML(XGBoost, test)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:07.885850Z","iopub.execute_input":"2024-09-24T12:45:07.886477Z","iopub.status.idle":"2024-09-24T12:45:23.059948Z","shell.execute_reply.started":"2024-09-24T12:45:07.886424Z","shell.execute_reply":"2024-09-24T12:45:23.058534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CatBoost\nCatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0)\nSubmission_CatBoost , k_cat= TrainML(CatBoost, test)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:23.061938Z","iopub.execute_input":"2024-09-24T12:45:23.062421Z","iopub.status.idle":"2024-09-24T12:45:32.064244Z","shell.execute_reply.started":"2024-09-24T12:45:23.062374Z","shell.execute_reply":"2024-09-24T12:45:32.062702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n\n# Params = {'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207, 'min_data_in_leaf': 41,\n#                 'feature_fraction': 0.6385678848225935, 'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6, \n#                             'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983} # LB : 0.452\n\n# Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n# Submission_LGBM = TrainML(Light,test)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.065906Z","iopub.execute_input":"2024-09-24T12:45:32.066465Z","iopub.status.idle":"2024-09-24T12:45:32.073190Z","shell.execute_reply.started":"2024-09-24T12:45:32.066402Z","shell.execute_reply":"2024-09-24T12:45:32.071525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SUBMIT","metadata":{}},{"cell_type":"code","source":"print(Submission_LGBM['sii'].value_counts())\nprint(Submission_XGB['sii'].value_counts())\nprint(Submission_CatBoost['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.074979Z","iopub.execute_input":"2024-09-24T12:45:32.075469Z","iopub.status.idle":"2024-09-24T12:45:32.094795Z","shell.execute_reply.started":"2024-09-24T12:45:32.075422Z","shell.execute_reply":"2024-09-24T12:45:32.093221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# k値の合計を計算して、各モデルの重みを計算\ntotal_k = k_cat + k_xgb + k_lgbm\n\nweight_cat = k_cat / total_k    # CatBoost の重み\nweight_xgb = k_xgb / total_k    # XGBoost の重み\nweight_lgbm = k_lgbm / total_k  # LightGBM の重み\n\n# 各モデルの予測結果（submission）を用意\n# 'sii' がカテゴリラベルであることを前提とします\nensemble_df = pd.DataFrame({\n    'id': Submission_LGBM['id'],\n    'cat': Submission_CatBoost[\"sii\"],\n    'xgb': Submission_XGB[\"sii\"],\n    'lgbm': Submission_LGBM[\"sii\"]\n})\n\n# 予測結果を長い形式に変換\nmelted = ensemble_df.melt(id_vars='id', value_vars=['cat', 'xgb', 'lgbm'], \n                          var_name='model', value_name='sii')\n\n# 各モデルに対応する重みを割り当て\nmelted['weight'] = melted['model'].map({\n    'cat': weight_cat,\n    'xgb': weight_xgb,\n    'lgbm': weight_lgbm\n})\n\n# 各idごと、siiごとに重みを集計\ngrouped = melted.groupby(['id', 'sii'])['weight'].sum().reset_index()\n\n# 各idごとに最大の重みを持つsiiを選択\nbest_submission = grouped.loc[grouped.groupby('id')['weight'].idxmax()][['id', 'sii']]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.096552Z","iopub.execute_input":"2024-09-24T12:45:32.097055Z","iopub.status.idle":"2024-09-24T12:45:32.130942Z","shell.execute_reply.started":"2024-09-24T12:45:32.096993Z","shell.execute_reply":"2024-09-24T12:45:32.129352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comparison_df = best_submission.merge(ensemble_df, on='id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.133106Z","iopub.execute_input":"2024-09-24T12:45:32.133701Z","iopub.status.idle":"2024-09-24T12:45:32.154168Z","shell.execute_reply.started":"2024-09-24T12:45:32.133629Z","shell.execute_reply":"2024-09-24T12:45:32.152588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comparison_df","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.156229Z","iopub.execute_input":"2024-09-24T12:45:32.156893Z","iopub.status.idle":"2024-09-24T12:45:32.180719Z","shell.execute_reply.started":"2024-09-24T12:45:32.156824Z","shell.execute_reply":"2024-09-24T12:45:32.179378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nbest_submission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-24T12:45:32.182672Z","iopub.execute_input":"2024-09-24T12:45:32.183246Z","iopub.status.idle":"2024-09-24T12:45:32.198876Z","shell.execute_reply.started":"2024-09-24T12:45:32.183196Z","shell.execute_reply":"2024-09-24T12:45:32.197218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}