{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"papermill":{"default_parameters":{},"duration":748.936636,"end_time":"2024-10-06T15:18:40.546997","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-06T15:06:11.610361","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Team 3b1b**\n","metadata":{"papermill":{"duration":0.008096,"end_time":"2024-10-06T15:06:14.591406","exception":false,"start_time":"2024-10-06T15:06:14.583310","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Please upvote and comment for any discussion point","metadata":{"papermill":{"duration":0.007082,"end_time":"2024-10-06T15:06:14.605889","exception":false,"start_time":"2024-10-06T15:06:14.598807","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\n\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"papermill":{"duration":5.514911,"end_time":"2024-10-06T15:06:20.128136","exception":false,"start_time":"2024-10-06T15:06:14.613225","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:34:38.058473Z","iopub.execute_input":"2024-10-12T21:34:38.059041Z","iopub.status.idle":"2024-10-12T21:34:38.075622Z","shell.execute_reply.started":"2024-10-12T21:34:38.058987Z","shell.execute_reply":"2024-10-12T21:34:38.074226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\n\n# Function to process files and load data (same as your original one)\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\n# Build the autoencoder model\ndef build_autoencoder(input_dim, encoding_dim):\n    input_layer = Input(shape=(input_dim,))\n    \n    # Encoder: compressing the input\n    encoded = Dense(encoding_dim, activation='relu')(input_layer)\n    \n    # Decoder: reconstructing the input\n    decoded = Dense(input_dim, activation='sigmoid')(encoded)\n    \n    # Autoencoder model\n    autoencoder = Model(inputs=input_layer, outputs=decoded)\n    \n    # Encoder model (for getting the compressed representation)\n    encoder = Model(inputs=input_layer, outputs=encoded)\n    \n    autoencoder.compile(optimizer=Adam(), loss='mse')\n    \n    return autoencoder, encoder\n\n# Function to perform dimensionality reduction using an autoencoder\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    \"\"\"\n    Perform dimensionality reduction using an Autoencoder.\n\n    Parameters:\n    df (pd.DataFrame): The input DataFrame with numerical features.\n    encoding_dim (int): The dimension of the encoded space.\n    epochs (int): Number of epochs to train the autoencoder.\n    batch_size (int): Size of the batches for training.\n\n    Returns:\n    pd.DataFrame: DataFrame containing the reduced-dimensional representation (encoding).\n    \"\"\"\n    \n    # Step 1: Standardize the data\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    # Step 2: Build the autoencoder\n    input_dim = df_scaled.shape[1]\n    autoencoder, encoder = build_autoencoder(input_dim, encoding_dim)\n    \n    # Step 3: Train the autoencoder\n    autoencoder.fit(df_scaled, df_scaled, epochs=epochs, batch_size=batch_size, shuffle=True, verbose=1)\n    \n    # Step 4: Get the encoded (reduced) representation\n    encoded_data = encoder.predict(df_scaled)\n    \n    # Step 5: Create a DataFrame for the encoded features\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i+1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\n# Load the data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Drop 'id' column for training\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\n# Perform autoencoder dimensionality reduction\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=50, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=50, epochs=100, batch_size=32)","metadata":{"papermill":{"duration":129.093056,"end_time":"2024-10-06T15:08:29.228723","exception":false,"start_time":"2024-10-06T15:06:20.135667","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:34:43.780952Z","iopub.execute_input":"2024-10-12T21:34:43.781428Z","iopub.status.idle":"2024-10-12T21:37:36.939093Z","shell.execute_reply.started":"2024-10-12T21:34:43.781381Z","shell.execute_reply":"2024-10-12T21:37:36.937474Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ts_encoded","metadata":{"papermill":{"duration":0.130858,"end_time":"2024-10-06T15:08:29.442483","exception":false,"start_time":"2024-10-06T15:08:29.311625","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:40:42.004235Z","iopub.execute_input":"2024-10-12T21:40:42.004870Z","iopub.status.idle":"2024-10-12T21:40:42.059808Z","shell.execute_reply.started":"2024-10-12T21:40:42.004789Z","shell.execute_reply":"2024-10-12T21:40:42.058375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_series_cols = train_ts_encoded.columns.tolist()\n#time_series_cols.remove(\"id\")","metadata":{"papermill":{"duration":0.108,"end_time":"2024-10-06T15:08:29.667773","exception":false,"start_time":"2024-10-06T15:08:29.559773","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:40:54.693273Z","iopub.execute_input":"2024-10-12T21:40:54.693911Z","iopub.status.idle":"2024-10-12T21:40:54.700622Z","shell.execute_reply.started":"2024-10-12T21:40:54.693856Z","shell.execute_reply":"2024-10-12T21:40:54.698955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded[\"id\"]=test_ts[\"id\"]","metadata":{"papermill":{"duration":0.089637,"end_time":"2024-10-06T15:08:29.858513","exception":false,"start_time":"2024-10-06T15:08:29.768876","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T22:05:59.105842Z","iopub.execute_input":"2024-10-12T22:05:59.106464Z","iopub.status.idle":"2024-10-12T22:05:59.124477Z","shell.execute_reply.started":"2024-10-12T22:05:59.106409Z","shell.execute_reply":"2024-10-12T22:05:59.123063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts_encoded","metadata":{"papermill":{"duration":0.138759,"end_time":"2024-10-06T15:08:30.071922","exception":false,"start_time":"2024-10-06T15:08:29.933163","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:40:57.118693Z","iopub.execute_input":"2024-10-12T21:40:57.119188Z","iopub.status.idle":"2024-10-12T21:40:57.198758Z","shell.execute_reply.started":"2024-10-12T21:40:57.119139Z","shell.execute_reply":"2024-10-12T21:40:57.197132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\ntest_ = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\n#test = pd.merge(test, test_ts, how=\"inner\", on='id')","metadata":{"papermill":{"duration":0.123745,"end_time":"2024-10-06T15:08:30.271814","exception":false,"start_time":"2024-10-06T15:08:30.148069","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:40:58.805970Z","iopub.execute_input":"2024-10-12T21:40:58.806461Z","iopub.status.idle":"2024-10-12T21:40:58.828735Z","shell.execute_reply.started":"2024-10-12T21:40:58.806413Z","shell.execute_reply":"2024-10-12T21:40:58.827504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.id","metadata":{"papermill":{"duration":0.089321,"end_time":"2024-10-06T15:08:30.527653","exception":false,"start_time":"2024-10-06T15:08:30.438332","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:00.149404Z","iopub.execute_input":"2024-10-12T21:41:00.149867Z","iopub.status.idle":"2024-10-12T21:41:00.160208Z","shell.execute_reply.started":"2024-10-12T21:41:00.149800Z","shell.execute_reply":"2024-10-12T21:41:00.158868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_","metadata":{"papermill":{"duration":0.356062,"end_time":"2024-10-06T15:08:30.959945","exception":false,"start_time":"2024-10-06T15:08:30.603883","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:01.939779Z","iopub.execute_input":"2024-10-12T21:41:01.940287Z","iopub.status.idle":"2024-10-12T21:41:02.349042Z","shell.execute_reply.started":"2024-10-12T21:41:01.940239Z","shell.execute_reply":"2024-10-12T21:41:02.347756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\n","metadata":{"papermill":{"duration":0.181953,"end_time":"2024-10-06T15:08:31.228831","exception":false,"start_time":"2024-10-06T15:08:31.046878","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:05.538797Z","iopub.execute_input":"2024-10-12T21:41:05.539304Z","iopub.status.idle":"2024-10-12T21:41:05.604364Z","shell.execute_reply.started":"2024-10-12T21:41:05.539254Z","shell.execute_reply":"2024-10-12T21:41:05.602938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    unique_values=sorted(unique_values)\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    #mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_train).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-10-12T21:41:09.172667Z","iopub.execute_input":"2024-10-12T21:41:09.173871Z","iopub.status.idle":"2024-10-12T21:41:09.235374Z","shell.execute_reply.started":"2024-10-12T21:41:09.173791Z","shell.execute_reply":"2024-10-12T21:41:09.233930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain.head()","metadata":{"papermill":{"duration":0.176428,"end_time":"2024-10-06T15:08:31.489735","exception":false,"start_time":"2024-10-06T15:08:31.313307","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:11.921769Z","iopub.execute_input":"2024-10-12T21:41:11.923152Z","iopub.status.idle":"2024-10-12T21:41:12.044353Z","shell.execute_reply.started":"2024-10-12T21:41:11.923088Z","shell.execute_reply":"2024-10-12T21:41:12.042561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"papermill":{"duration":0.095925,"end_time":"2024-10-06T15:08:31.673323","exception":false,"start_time":"2024-10-06T15:08:31.577398","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:18.303317Z","iopub.execute_input":"2024-10-12T21:41:18.304572Z","iopub.status.idle":"2024-10-12T21:41:18.319805Z","shell.execute_reply.started":"2024-10-12T21:41:18.304503Z","shell.execute_reply":"2024-10-12T21:41:18.317952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntest.shape","metadata":{"papermill":{"duration":0.100032,"end_time":"2024-10-06T15:08:31.871701","exception":false,"start_time":"2024-10-06T15:08:31.771669","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:29.585811Z","iopub.execute_input":"2024-10-12T21:41:29.587349Z","iopub.status.idle":"2024-10-12T21:41:29.603983Z","shell.execute_reply.started":"2024-10-12T21:41:29.587285Z","shell.execute_reply":"2024-10-12T21:41:29.602049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission = TrainML(voting_model, test)\n\n# Save submission\n#Submission.to_csv('submission.csv', index=False)\n#print(Submission['sii'].value_counts())","metadata":{"papermill":{"duration":70.131528,"end_time":"2024-10-06T15:09:42.100925","exception":false,"start_time":"2024-10-06T15:08:31.969397","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:41:33.349019Z","iopub.execute_input":"2024-10-12T21:41:33.349464Z","iopub.status.idle":"2024-10-12T21:42:51.012337Z","shell.execute_reply.started":"2024-10-12T21:41:33.349418Z","shell.execute_reply":"2024-10-12T21:42:51.010080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:09:42.271321Z","iopub.status.busy":"2024-10-06T15:09:42.270880Z","iopub.status.idle":"2024-10-06T15:09:42.282904Z","shell.execute_reply":"2024-10-06T15:09:42.281556Z"},"papermill":{"duration":0.099557,"end_time":"2024-10-06T15:09:42.285321","exception":false,"start_time":"2024-10-06T15:09:42.185764","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\n\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\n\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    unique_values=sorted(unique_values)\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    #mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission1 = TrainML(voting_model, test)\n\n# Save submission\n#Submission1.to_csv('submission.csv', index=False)\n","metadata":{"papermill":{"duration":233.544604,"end_time":"2024-10-06T15:13:35.915801","exception":false,"start_time":"2024-10-06T15:09:42.371197","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:46:32.759223Z","iopub.execute_input":"2024-10-12T21:46:32.759771Z","iopub.status.idle":"2024-10-12T21:50:45.456118Z","shell.execute_reply.started":"2024-10-12T21:46:32.759713Z","shell.execute_reply":"2024-10-12T21:50:45.454556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission1","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:13:36.087269Z","iopub.status.busy":"2024-10-06T15:13:36.086383Z","iopub.status.idle":"2024-10-06T15:13:36.098226Z","shell.execute_reply":"2024-10-06T15:13:36.097018Z"},"papermill":{"duration":0.100641,"end_time":"2024-10-06T15:13:36.100592","exception":false,"start_time":"2024-10-06T15:13:35.999951","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.084942,"end_time":"2024-10-06T15:13:36.269649","exception":false,"start_time":"2024-10-06T15:13:36.184707","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score',\n                'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n                'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n                'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season',\n                'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',\n                'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n         'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    unique_values=sorted(unique_values)\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    #mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\n\n# Imputation step: Filling missing values with the median\nimputer = SimpleImputer(strategy='median')\n\n# Updating the ensemble to include the RandomForest and GradientBoosting models\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\n# Train the ensemble with the updated model pipeline\npredictions = TrainML(ensemble, test)\n\n# Save predictions to a CSV file\nsample['sii'] = predictions\n#sample.to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":300.119902,"end_time":"2024-10-06T15:18:36.476405","exception":false,"start_time":"2024-10-06T15:13:36.356503","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-10-12T21:51:13.446403Z","iopub.execute_input":"2024-10-12T21:51:13.446963Z","iopub.status.idle":"2024-10-12T21:56:40.361144Z","shell.execute_reply.started":"2024-10-12T21:51:13.446910Z","shell.execute_reply":"2024-10-12T21:56:40.359808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:36.670137Z","iopub.status.busy":"2024-10-06T15:18:36.669190Z","iopub.status.idle":"2024-10-06T15:18:36.683196Z","shell.execute_reply":"2024-10-06T15:18:36.681926Z"},"papermill":{"duration":0.112176,"end_time":"2024-10-06T15:18:36.685706","exception":false,"start_time":"2024-10-06T15:18:36.573530","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load your three submission files\nsub1 = Submission\nsub2 = Submission1\nsub3 = sample\n\n# Ensure the IDs are aligned (if not sorted)\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\n# Combine the three predictions\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\n# Apply majority voting\ndef majority_vote(row):\n    return row.mode()[0]  # Mode gets the most frequent value (majority vote)\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\n# Create the final submission DataFrame\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n# Save the final submission to a CSV file\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:36.862491Z","iopub.status.busy":"2024-10-06T15:18:36.862058Z","iopub.status.idle":"2024-10-06T15:18:36.894929Z","shell.execute_reply":"2024-10-06T15:18:36.893365Z"},"papermill":{"duration":0.12448,"end_time":"2024-10-06T15:18:36.897882","exception":false,"start_time":"2024-10-06T15:18:36.773402","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:37.072959Z","iopub.status.busy":"2024-10-06T15:18:37.072515Z","iopub.status.idle":"2024-10-06T15:18:37.084993Z","shell.execute_reply":"2024-10-06T15:18:37.083626Z"},"papermill":{"duration":0.103253,"end_time":"2024-10-06T15:18:37.087502","exception":false,"start_time":"2024-10-06T15:18:36.984249","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = train.drop(['sii'], axis=1)\n# y = train['sii']","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:37.269221Z","iopub.status.busy":"2024-10-06T15:18:37.268781Z","iopub.status.idle":"2024-10-06T15:18:37.273838Z","shell.execute_reply":"2024-10-06T15:18:37.272546Z"},"papermill":{"duration":0.094783,"end_time":"2024-10-06T15:18:37.276375","exception":false,"start_time":"2024-10-06T15:18:37.181592","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train=X\n# y_train=y","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:37.448936Z","iopub.status.busy":"2024-10-06T15:18:37.448515Z","iopub.status.idle":"2024-10-06T15:18:37.455161Z","shell.execute_reply":"2024-10-06T15:18:37.453891Z"},"papermill":{"duration":0.09573,"end_time":"2024-10-06T15:18:37.457842","exception":false,"start_time":"2024-10-06T15:18:37.362112","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import RandomizedSearchCV\n# from sklearn.impute import SimpleImputer\n# from sklearn.pipeline import Pipeline\n# from sklearn.ensemble import VotingRegressor\n# from lightgbm import LGBMRegressor\n# from xgboost import XGBRegressor\n# from catboost import CatBoostRegressor\n# from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\n\n# # Define hyperparameter grids for each model\n\n# # LightGBM hyperparameter grid\n# lgb_param_grid = {\n#     'regressor__learning_rate': [0.01, 0.03, 0.046, 0.1],\n#     'regressor__max_depth': [8, 12, 16],\n#     'regressor__num_leaves': [128, 256, 478, 512],\n#     'regressor__min_data_in_leaf': [10, 13, 16],\n#     'regressor__feature_fraction': [0.7, 0.8, 0.893],\n#     'regressor__bagging_fraction': [0.7, 0.784, 0.85],\n#     'regressor__lambda_l1': [5, 10, 15],\n#     'regressor__lambda_l2': [0.001, 0.01, 0.1]\n# }\n\n# # XGBoost hyperparameter grid\n# xgb_param_grid = {\n#     'regressor__learning_rate': [0.01, 0.05, 0.1],\n#     'regressor__max_depth': [4, 6, 8],\n#     'regressor__n_estimators': [100, 200, 300],\n#     'regressor__subsample': [0.7, 0.8, 0.9],\n#     'regressor__colsample_bytree': [0.7, 0.8, 0.9],\n#     'regressor__reg_alpha': [0.1, 1, 5],\n#     'regressor__reg_lambda': [1, 5, 10]\n# }\n\n# # CatBoost hyperparameter grid\n# cat_param_grid = {\n#     'regressor__learning_rate': [0.03, 0.05, 0.07],\n#     'regressor__depth': [4, 6, 8],\n#     'regressor__iterations': [200, 300, 400],\n#     'regressor__l2_leaf_reg': [3, 10, 15]\n# }\n\n# # RandomForest hyperparameter grid\n# rf_param_grid = {\n#     'regressor__n_estimators': [100, 200, 300],\n#     'regressor__max_depth': [10, 20, 30],\n#     'regressor__min_samples_split': [2, 5, 10],\n#     'regressor__min_samples_leaf': [1, 2, 4]\n# }\n\n# # GradientBoosting hyperparameter grid\n# gb_param_grid = {\n#     'regressor__learning_rate': [0.01, 0.05, 0.1],\n#     'regressor__n_estimators': [100, 200, 300],\n#     'regressor__max_depth': [3, 5, 7],\n#     'regressor__subsample': [0.7, 0.8, 0.9]\n# }\n\n# # Imputer for handling missing values\n# imputer = SimpleImputer(strategy='median')\n\n# # Pipelines for each regressor\n# lgb_pipeline = Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])\n# xgb_pipeline = Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])\n# cat_pipeline = Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])\n# rf_pipeline = Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])\n# gb_pipeline = Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])\n\n# # Perform RandomizedSearchCV for each model\n\n# # LightGBM\n# lgb_search = RandomizedSearchCV(lgb_pipeline, lgb_param_grid, n_iter=20, scoring='neg_mean_squared_error', cv=3, random_state=SEED)\n# lgb_search.fit(X_train, y_train)\n\n# # XGBoost\n# xgb_search = RandomizedSearchCV(xgb_pipeline, xgb_param_grid, n_iter=20, scoring='neg_mean_squared_error', cv=3, random_state=SEED)\n# xgb_search.fit(X_train, y_train)\n\n# # CatBoost\n# cat_search = RandomizedSearchCV(cat_pipeline, cat_param_grid, n_iter=20, scoring='neg_mean_squared_error', cv=3, random_state=SEED)\n# cat_search.fit(X_train, y_train)\n\n# # RandomForest\n# rf_search = RandomizedSearchCV(rf_pipeline, rf_param_grid, n_iter=20, scoring='neg_mean_squared_error', cv=3, random_state=SEED)\n# rf_search.fit(X_train, y_train)\n\n# # GradientBoosting\n# gb_search = RandomizedSearchCV(gb_pipeline, gb_param_grid, n_iter=20, scoring='neg_mean_squared_error', cv=3, random_state=SEED)\n# gb_search.fit(X_train, y_train)\n\n# # Get the best models from the RandomizedSearchCV results\n# best_lgb = lgb_search.best_estimator_\n# best_xgb = xgb_search.best_estimator_\n# best_cat = cat_search.best_estimator_\n# best_rf = rf_search.best_estimator_\n# best_gb = gb_search.best_estimator_\n\n# # Create a new ensemble with the best estimators\n# ensemble = VotingRegressor(estimators=[\n#     ('lgb', best_lgb),\n#     ('xgb', best_xgb),\n#     ('cat', best_cat),\n#     ('rf', best_rf),\n#     ('gb', best_gb)\n# ])\n\n# # Train the ensemble\n# predictions = TrainML(ensemble, test)\n\n# # Save predictions to a CSV file\n# sample['sii'] = predictions\n# sample.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2024-10-06T15:18:37.637047Z","iopub.status.busy":"2024-10-06T15:18:37.636590Z","iopub.status.idle":"2024-10-06T15:18:37.646935Z","shell.execute_reply":"2024-10-06T15:18:37.645122Z"},"papermill":{"duration":0.104202,"end_time":"2024-10-06T15:18:37.649867","exception":false,"start_time":"2024-10-06T15:18:37.545665","status":"completed"},"tags":[],"collapsed":true,"jupyter":{"outputs_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.087125,"end_time":"2024-10-06T15:18:37.825984","exception":false,"start_time":"2024-10-06T15:18:37.738859","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}