{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom sklearn.base import clone\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.pipeline import Pipeline\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nfrom concurrent.futures import ThreadPoolExecutor\n\n# Suppress warnings and set display options\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\n# Set random seed for reproducibility\nSEED = 30\nnp.random.seed(SEED)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:36.437890Z","iopub.execute_input":"2024-12-13T17:06:36.438299Z","iopub.status.idle":"2024-12-13T17:06:36.445856Z","shell.execute_reply.started":"2024-12-13T17:06:36.438262Z","shell.execute_reply":"2024-12-13T17:06:36.444601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Display the shapes of the train and test datasets\nprint(f'Train shape: {train.shape}')\nprint(f'Test shape: {test.shape}')\n\n# Identify columns missing in the test set\nmissing_in_test = [col for col in train.columns if col not in test.columns and col not in ['id','sii']] #We know sii value is missing (test), and ID is not needed\nprint('Columns missing in test:', missing_in_test) #Should def clean this up a little bit lol \n\n# Drop the missing columns from the training data (they are not used in test so why add them?)\ntrain = train.drop(columns=missing_in_test)\nprint(f'Train shape after dropping missing columns: {train.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:36.713679Z","iopub.execute_input":"2024-12-13T17:06:36.714079Z","iopub.status.idle":"2024-12-13T17:06:36.767312Z","shell.execute_reply.started":"2024-12-13T17:06:36.714043Z","shell.execute_reply":"2024-12-13T17:06:36.766157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify categorical and numerical columns\ncategorical_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Sex', 'CGAS-Season', 'Physical-Season', \n                    'Fitness_Endurance-Season', 'FGC-Season', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', \n                    'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', \n                    'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num', \n                    'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season', \n                    'PreInt_EduHx-computerinternet_hoursday'] #only using these from .dic all categorical features besides ones that got dropped (PCIAT-Season is numeric but not in testing data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:36.986452Z","iopub.execute_input":"2024-12-13T17:06:36.986861Z","iopub.status.idle":"2024-12-13T17:06:36.994760Z","shell.execute_reply.started":"2024-12-13T17:06:36.986824Z","shell.execute_reply":"2024-12-13T17:06:36.993675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def process_file(filename, dirname):\n#     df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n#     df.drop('step', axis=1, inplace=True)\n#     return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# def load_time_series(dirname) -> pd.DataFrame:\n#     ids = os.listdir(dirname)\n    \n#     with ThreadPoolExecutor() as executor:\n#         results = list(tqdm(\n#             executor.map(lambda fname: process_file(fname, dirname), ids),\n#             total=len(ids))\n#         )\n    \n#     stats, indexes = zip(*results)\n    \n#     # Create generic column names for all statistics\n#     df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n#     df['id'] = indexes\n    \n#     return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:37.278710Z","iopub.execute_input":"2024-12-13T17:06:37.279096Z","iopub.status.idle":"2024-12-13T17:06:37.284974Z","shell.execute_reply.started":"2024-12-13T17:06:37.279062Z","shell.execute_reply":"2024-12-13T17:06:37.283690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# base_path = '/kaggle/input/child-mind-institute-problematic-internet-use'\n# train_parquet_path = f'{base_path}/series_train.parquet'\n# test_parquet_path = f'{base_path}/series_test.parquet'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:37.551079Z","iopub.execute_input":"2024-12-13T17:06:37.551503Z","iopub.status.idle":"2024-12-13T17:06:37.559111Z","shell.execute_reply.started":"2024-12-13T17:06:37.551460Z","shell.execute_reply":"2024-12-13T17:06:37.558007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"\\nProcessing parquet files...\")\n# try:\n#     print(\"Loading and processing training data...\")\n#     train_ts = load_time_series(train_parquet_path)\n#     print(\"\\nShape of training features:\", train_ts.shape)\n    \n#     print(\"\\nLoading and processing test data...\")\n#     test_ts = load_time_series(test_parquet_path)\n#     print(\"\\nShape of test features:\", test_ts.shape)\n    \n#     # Save features\n#     print(\"\\nSaving features...\")\n#     train_ts.to_parquet(\"actigraphy_features_train.parquet\")\n#     test_ts.to_parquet(\"actigraphy_features_test.parquet\")\n# except Exception as e:\n#     print(f\"Error processing parquet files: {str(e)}\")\n#     raise","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:37.825783Z","iopub.execute_input":"2024-12-13T17:06:37.826165Z","iopub.status.idle":"2024-12-13T17:06:37.831739Z","shell.execute_reply.started":"2024-12-13T17:06:37.826131Z","shell.execute_reply":"2024-12-13T17:06:37.830522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"\\nLoading parquet data...\")\n# try:\n#     train_ts = pd.read_parquet(\"actigraphy_features_train.parquet\")\n#     test_ts = pd.read_parquet(\"actigraphy_features_test.parquet\")\n# except Exception as e:\n#     print(f\"Error in loading processed parquet data: {str(e)}\")\n#     raise","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:38.105669Z","iopub.execute_input":"2024-12-13T17:06:38.106061Z","iopub.status.idle":"2024-12-13T17:06:38.110640Z","shell.execute_reply.started":"2024-12-13T17:06:38.106027Z","shell.execute_reply":"2024-12-13T17:06:38.109547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Merge parquet features with original data\n# print(\"\\nMerging parquet data with CSV data...\")\n# try:\n#     train = pd.merge(train, train_ts, how=\"left\", on='id')\n#     test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n#     print(\"\\nInitial shapes:\")\n#     print(\"Train combined:\", train.shape)\n#     print(\"Test combined:\", test.shape)\n# except Exception as e:\n#     print(f\"Error in merging: {str(e)}\")\n#     raise","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:38.380040Z","iopub.execute_input":"2024-12-13T17:06:38.380436Z","iopub.status.idle":"2024-12-13T17:06:38.385295Z","shell.execute_reply.started":"2024-12-13T17:06:38.380401Z","shell.execute_reply":"2024-12-13T17:06:38.384258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the features by excluding the missing columns and identifiers\nfeatures = [col for col in train.columns if col not in ['id', 'sii']]\nprint(f\"Number of features used: {len(features)}\")\n\nnumerical_cols = [col for col in features if col not in categorical_cols]\nprint(f'Categorical columns: {categorical_cols}')\nprint(f'Numerical columns: {numerical_cols[:100]}... (truncated)')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:38.660519Z","iopub.execute_input":"2024-12-13T17:06:38.660941Z","iopub.status.idle":"2024-12-13T17:06:38.669893Z","shell.execute_reply.started":"2024-12-13T17:06:38.660905Z","shell.execute_reply":"2024-12-13T17:06:38.668663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing function\ndef preprocess_data(train_df, test_df, categorical_cols):\n    \n    train_df = train_df.copy()\n    test_df = test_df.copy()\n\n    # Label Encode categorical features\n    for col in categorical_cols:\n        le = LabelEncoder()\n        combined = pd.concat([train_df[col], test_df[col]]).astype(str)\n        le.fit(combined)\n        train_df[col] = le.transform(train_df[col].astype(str))\n        test_df[col] = le.transform(test_df[col].astype(str))\n\n    return train_df, test_df\n\n# Apply preprocessing\ntrain_processed, test_processed = preprocess_data(train, test, categorical_cols)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:39.233628Z","iopub.execute_input":"2024-12-13T17:06:39.234018Z","iopub.status.idle":"2024-12-13T17:06:39.344325Z","shell.execute_reply.started":"2024-12-13T17:06:39.233981Z","shell.execute_reply":"2024-12-13T17:06:39.343480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed, test_processed = preprocess_data(train, test, categorical_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:39.714777Z","iopub.execute_input":"2024-12-13T17:06:39.715172Z","iopub.status.idle":"2024-12-13T17:06:39.825062Z","shell.execute_reply.started":"2024-12-13T17:06:39.715136Z","shell.execute_reply":"2024-12-13T17:06:39.824216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering function\ndef feature_engineering(df):\n    \n    # df = df.copy()\n    # season_cols = [col for col in df.columns if 'Season' in col]\n    # df = df.drop(season_cols, axis=1) \n    df['BMI_by_Age'] = df['Physical-BMI'] * (df['Basic_Demos-Age'])\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['HeartRate_BP_Ratio'] = df['Physical-HeartRate'] / (df['Physical-Systolic_BP'])\n    \n    # df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    # df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    # df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    # df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    # df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    # df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    # df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    # df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    # df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    # df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    # df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    # df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    # df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    # df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    # df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:40.123192Z","iopub.execute_input":"2024-12-13T17:06:40.123550Z","iopub.status.idle":"2024-12-13T17:06:40.130108Z","shell.execute_reply.started":"2024-12-13T17:06:40.123519Z","shell.execute_reply":"2024-12-13T17:06:40.128848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# season_cols = [col for col in categorical_cols if 'Season' in col]\n# print(\"Season columns to be removed:\", season_cols)  # Let's verify what we're removing\n\n# Remove Season columns\n# train_processed = train_processed.drop(season_cols, axis=1)\n# test_processed = test_processed.drop(season_cols, axis=1)\n\n# engineered_features = ['HeartRate_BP_Ratio','BMI_Age','Internet_Hours_Age','BMI_Internet_Hours','BFP_BMI','FMI_BFP',\n#                        'LST_TBW','BFP_BMR','BFP_DEE','BMR_Weight','DEE_Weight','SMM_Height', 'Muscle_to_Fat',\n#                       'Hydration_Status','ICW_TBW']\n\nengineered_features = ['BMI_by_Age','Internet_Hours_Age','HeartRate_BP_Ratio']\n\n\n\nreduced_features = [col for col in train_processed.columns if col not in ['id', 'sii']]\n\nfinal_cols =  reduced_features + engineered_features\n\n# Apply feature engineering\ntrain_processed = feature_engineering(train_processed)\ntest_processed = feature_engineering(test_processed)\n\n# Remove any rows in training data where 'sii' is NaN\ntrain_processed = train_processed[train_processed['sii'].notna()].reset_index(drop=True)\n\n# Define feature matrix X and target vector y\n# X = train_processed[features + ['BMI_by_Age','Internet_Hours_Age','HeartRate_BP_Ratio']]\n\nX = train_processed[final_cols]\n\ny = train_processed['sii'].astype(int)\n\n# Define test feature matrix\n# X_test = test_processed[features + ['BMI_by_Age','Internet_Hours_Age','HeartRate_BP_Ratio']]\nX_test = test_processed[final_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:40.477516Z","iopub.execute_input":"2024-12-13T17:06:40.477960Z","iopub.status.idle":"2024-12-13T17:06:40.498007Z","shell.execute_reply.started":"2024-12-13T17:06:40.477922Z","shell.execute_reply":"2024-12-13T17:06:40.497007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def remove_correlated_features_split(X_train, X_test, \n#                                    parquet_threshold,\n#                                    original_threshold):\n    \n#     # Split features into parquet and original\n#     parquet_cols = [col for col in X_train.columns if col.startswith('Stat_')]\n#     original_cols = [col for col in X_train.columns if not col.startswith('Stat_')]\n    \n#     print(f\"Initial parquet features: {len(parquet_cols)}\")\n#     print(f\"Initial original features: {len(original_cols)}\")\n    \n#     # Process parquet features\n#     parquet_corr = X_train[parquet_cols].corr().abs()\n#     upper_parquet = parquet_corr.where(np.triu(np.ones(parquet_corr.shape), k=1).astype(bool))\n#     to_drop_parquet = [c for c in upper_parquet.columns if any(upper_parquet[c] > parquet_threshold)]\n    \n#     # Process original features\n#     original_corr = X_train[original_cols].corr().abs()\n#     upper_original = original_corr.where(np.triu(np.ones(original_corr.shape), k=1).astype(bool))\n#     to_drop_original = [c for c in upper_original.columns if any(upper_original[c] > original_threshold)]\n    \n#     # Remove correlated features\n#     X_train_reduced = X_train.drop(columns=to_drop_parquet + to_drop_original)\n#     X_test_reduced = X_test.drop(columns=to_drop_parquet + to_drop_original)\n    \n#     print(f\"\\nFeatures removed:\")\n#     print(f\"Parquet features removed: {len(to_drop_parquet)}\")\n#     print(f\"Original features removed: {len(to_drop_original)}\")\n#     print(f\"Final feature count: {X_train_reduced.shape[1]}\")\n    \n#     return X_train_reduced, X_test_reduced\n\n\n# X_reduced, X_test_reduced = remove_correlated_features_split(\n#     X, X_test,\n#     parquet_threshold=0.50,  \n#     original_threshold=0.97\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:40.812727Z","iopub.execute_input":"2024-12-13T17:06:40.813120Z","iopub.status.idle":"2024-12-13T17:06:40.819335Z","shell.execute_reply.started":"2024-12-13T17:06:40.813084Z","shell.execute_reply":"2024-12-13T17:06:40.818196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_reduced = X\nX_test_reduced = X_test\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:41.909740Z","iopub.execute_input":"2024-12-13T17:06:41.910123Z","iopub.status.idle":"2024-12-13T17:06:41.916893Z","shell.execute_reply.started":"2024-12-13T17:06:41.910089Z","shell.execute_reply":"2024-12-13T17:06:41.915632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the Quadratic Weighted Kappa metric\ndef quadratic_weighted_kappa(y_true, y_pred):\n    y_pred = np.round(np.clip(y_pred, 0, 3)).astype(int)\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Function to optimize thresholds for QWK\ndef optimize_thresholds(y_true, y_pred):\n    def loss_function(thresholds):\n        thresholds = np.sort(thresholds)\n        y_pred_adj = np.digitize(y_pred, bins=thresholds)\n        return -quadratic_weighted_kappa(y_true, y_pred_adj)\n    \n    initial_thresholds = [0.5, 1.5, 2.5] \n    \n#Predictions ≤ 0.5: Class 0\n#0.5 < Predictions ≤ 1.5: Class 1\n#1.5 < Predictions ≤ 2.5: Class 2\n#Predictions > 2.5: Class 3\n    \n    result = minimize(loss_function, initial_thresholds, method='nelder-mead') #thresholds that minimize the loss function for best score qwk\n    return result.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:42.732302Z","iopub.execute_input":"2024-12-13T17:06:42.732847Z","iopub.status.idle":"2024-12-13T17:06:42.740615Z","shell.execute_reply.started":"2024-12-13T17:06:42.732791Z","shell.execute_reply":"2024-12-13T17:06:42.739268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NEW\nSEED = 30\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 5,  \n    'lambda_l2': 0.01,  \n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  \n    'reg_lambda': 5,  \n    'random_state': SEED,\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Define cross-validation strategy\nn_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# Initialize arrays to store out-of-fold and test predictions\noof_predictions = np.zeros(len(X_reduced))\ntest_predictions = np.zeros(len(X_test_reduced))\n\n# Combine models using Voting Regressor\nensemble_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Perform cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_reduced, y)):\n    print(f'Fold {fold+1}')\n    X_train, X_val = X_reduced.iloc[train_idx], X_reduced.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # Clone the ensemble model to ensure fresh weights for each fold\n    model = clone(ensemble_model)\n    model.fit(X_train, y_train)\n    \n    # Predict on validation and test sets\n    val_pred = model.predict(X_val)\n    oof_predictions[val_idx] = val_pred\n    \n    test_pred = model.predict(X_test_reduced)\n    test_predictions += test_pred / n_splits\n    \n    # Calculate and print QWK for the current fold\n    val_kappa = quadratic_weighted_kappa(y_val, val_pred)\n    print(f'Fold {fold+1} QWK: {val_kappa:.4f}')\n\n\n\n#Optimize thresholds based on out-of-fold predictions\noptimal_thresholds = optimize_thresholds(y, oof_predictions)\nprint(f'Optimal thresholds: {optimal_thresholds}')\n\n# Adjust out-of-fold predictions using optimized thresholds and calculate final QWK\noof_pred_adj = np.digitize(oof_predictions, bins=optimal_thresholds)\nfinal_kappa = quadratic_weighted_kappa(y, oof_pred_adj)\nprint(f'Final QWK on OOF predictions: {final_kappa:.4f}')\n\n# Adjust test predictions using optimized thresholds\ntest_pred_adj = np.digitize(test_predictions, bins=optimal_thresholds)\ntest_pred_adj = np.clip(test_pred_adj, 0, 3).astype(int)\n\n# Prepare the submission DataFrame\nSubmission1 = pd.DataFrame({\n    'id': test['id'],\n    'sii': test_pred_adj\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:06:43.262637Z","iopub.execute_input":"2024-12-13T17:06:43.263035Z","iopub.status.idle":"2024-12-13T17:07:08.558498Z","shell.execute_reply.started":"2024-12-13T17:06:43.262998Z","shell.execute_reply":"2024-12-13T17:07:08.557299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Define the ensemble model using VotingRegressor\n# ensemble_model = VotingRegressor(estimators=[\n#     ('lgbm', LGBMRegressor(n_estimators=200, learning_rate=0.05, random_state=SEED)),\n#     ('xgb', XGBRegressor(n_estimators=200, learning_rate=0.05, random_state=SEED, verbosity=0)),\n#     ('cat', CatBoostRegressor(n_estimators=200, learning_rate=0.05, random_state=SEED, silent=True))\n# ])\n\n# # Define cross-validation strategy\n# n_splits = 5\n# kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# # Initialize arrays to store out-of-fold and test predictions\n# oof_predictions = np.zeros(len(X_reduced))\n# test_predictions = np.zeros(len(X_test_reduced))\n\n# # Perform cross-validation\n# for fold, (train_idx, val_idx) in enumerate(kf.split(X_reduced, y)):\n#     print(f'Fold {fold+1}')\n#     X_train, X_val = X_reduced.iloc[train_idx], X_reduced.iloc[val_idx]\n#     y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n#     # Clone the ensemble model to ensure fresh weights for each fold\n#     model = clone(ensemble_model)\n#     model.fit(X_train, y_train)\n    \n#     # Predict on validation and test sets\n#     val_pred = model.predict(X_val)\n#     oof_predictions[val_idx] = val_pred\n    \n#     test_pred = model.predict(X_test_reduced)\n#     test_predictions += test_pred / n_splits\n    \n#     # Calculate and print QWK for the current fold\n#     val_kappa = quadratic_weighted_kappa(y_val, val_pred)\n#     print(f'Fold {fold+1} QWK: {val_kappa:.4f}')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:07:08.560732Z","iopub.execute_input":"2024-12-13T17:07:08.561601Z","iopub.status.idle":"2024-12-13T17:07:08.568125Z","shell.execute_reply.started":"2024-12-13T17:07:08.561526Z","shell.execute_reply":"2024-12-13T17:07:08.566780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Optimize thresholds based on out-of-fold predictions\n# optimal_thresholds = optimize_thresholds(y, oof_predictions)\n# print(f'Optimal thresholds: {optimal_thresholds}')\n\n# # Adjust out-of-fold predictions using optimized thresholds and calculate final QWK\n# oof_pred_adj = np.digitize(oof_predictions, bins=optimal_thresholds)\n# final_kappa = quadratic_weighted_kappa(y, oof_pred_adj)\n# print(f'Final QWK on OOF predictions: {final_kappa:.4f}')\n\n# # Adjust test predictions using optimized thresholds\n# test_pred_adj = np.digitize(test_predictions, bins=optimal_thresholds)\n# test_pred_adj = np.clip(test_pred_adj, 0, 3).astype(int)\n\n# # Prepare the submission DataFrame\n# Submission1 = pd.DataFrame({\n#     'id': test['id'],\n#     'sii': test_pred_adj\n# })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:07:08.569628Z","iopub.execute_input":"2024-12-13T17:07:08.570051Z","iopub.status.idle":"2024-12-13T17:07:08.584289Z","shell.execute_reply.started":"2024-12-13T17:07:08.570002Z","shell.execute_reply":"2024-12-13T17:07:08.583176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 1\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n\n# Define cross-validation strategy\nn_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# Initialize arrays to store out-of-fold and test predictions\noof_predictions = np.zeros(len(X_reduced))\ntest_predictions = np.zeros(len(X_test_reduced))\n\n# Perform cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_reduced, y)):\n    print(f'Fold {fold+1}')\n    X_train, X_val = X_reduced.iloc[train_idx], X_reduced.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # Clone the ensemble model to ensure fresh weights for each fold\n    model = clone(ensemble_model)\n    model.fit(X_train, y_train)\n    \n    # Predict on validation and test sets\n    val_pred = model.predict(X_val)\n    oof_predictions[val_idx] = val_pred\n    \n    test_pred = model.predict(X_test_reduced)\n    test_predictions += test_pred / n_splits\n    \n    # Calculate and print QWK for the current fold\n    val_kappa = quadratic_weighted_kappa(y_val, val_pred)\n    print(f'Fold {fold+1} QWK: {val_kappa:.4f}')\n\n\n\n# Combine models using Voting Regressor\nensemble_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n#Optimize thresholds based on out-of-fold predictions\noptimal_thresholds = optimize_thresholds(y, oof_predictions)\nprint(f'Optimal thresholds: {optimal_thresholds}')\n\n# Adjust out-of-fold predictions using optimized thresholds and calculate final QWK\noof_pred_adj = np.digitize(oof_predictions, bins=optimal_thresholds)\nfinal_kappa = quadratic_weighted_kappa(y, oof_pred_adj)\nprint(f'Final QWK on OOF predictions: {final_kappa:.4f}')\n\n# Adjust test predictions using optimized thresholds\ntest_pred_adj = np.digitize(test_predictions, bins=optimal_thresholds)\ntest_pred_adj = np.clip(test_pred_adj, 0, 3).astype(int)\n\n# Prepare the submission DataFrame\nSubmission2 = pd.DataFrame({\n    'id': test['id'],\n    'sii': test_pred_adj\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:07:08.587269Z","iopub.execute_input":"2024-12-13T17:07:08.588446Z","iopub.status.idle":"2024-12-13T17:07:34.063304Z","shell.execute_reply.started":"2024-12-13T17:07:08.588390Z","shell.execute_reply":"2024-12-13T17:07:34.062314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 100\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Define cross-validation strategy\nn_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# Initialize arrays to store out-of-fold and test predictions\noof_predictions = np.zeros(len(X_reduced))\ntest_predictions = np.zeros(len(X_test_reduced))\n\n# Perform cross-validation\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_reduced, y)):\n    print(f'Fold {fold+1}')\n    X_train, X_val = X_reduced.iloc[train_idx], X_reduced.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # Clone the ensemble model to ensure fresh weights for each fold\n    model = clone(ensemble_model)\n    model.fit(X_train, y_train)\n    \n    # Predict on validation and test sets\n    val_pred = model.predict(X_val)\n    oof_predictions[val_idx] = val_pred\n    \n    test_pred = model.predict(X_test_reduced)\n    test_predictions += test_pred / n_splits\n    \n    # Calculate and print QWK for the current fold\n    val_kappa = quadratic_weighted_kappa(y_val, val_pred)\n    print(f'Fold {fold+1} QWK: {val_kappa:.4f}')\n\n\n\n# Combine models using Voting Regressor\nensemble_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n#Optimize thresholds based on out-of-fold predictions\noptimal_thresholds = optimize_thresholds(y, oof_predictions)\nprint(f'Optimal thresholds: {optimal_thresholds}')\n\n# Adjust out-of-fold predictions using optimized thresholds and calculate final QWK\noof_pred_adj = np.digitize(oof_predictions, bins=optimal_thresholds)\nfinal_kappa = quadratic_weighted_kappa(y, oof_pred_adj)\nprint(f'Final QWK on OOF predictions: {final_kappa:.4f}')\n\n# Adjust test predictions using optimized thresholds\ntest_pred_adj = np.digitize(test_predictions, bins=optimal_thresholds)\ntest_pred_adj = np.clip(test_pred_adj, 0, 3).astype(int)\n\n# Prepare the submission DataFrame\nSubmission3 = pd.DataFrame({\n    'id': test['id'],\n    'sii': test_pred_adj\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:07:34.064680Z","iopub.execute_input":"2024-12-13T17:07:34.065060Z","iopub.status.idle":"2024-12-13T17:07:56.590419Z","shell.execute_reply.started":"2024-12-13T17:07:34.065025Z","shell.execute_reply":"2024-12-13T17:07:56.589379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T17:07:56.592061Z","iopub.execute_input":"2024-12-13T17:07:56.592898Z","iopub.status.idle":"2024-12-13T17:07:56.615520Z","shell.execute_reply.started":"2024-12-13T17:07:56.592846Z","shell.execute_reply":"2024-12-13T17:07:56.614386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}