{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9766817,"sourceType":"datasetVersion","datasetId":5981682}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pyarrow.parquet as pq\n\nimport torch\n\nimport torch.nn as nn\n\nfrom torch.utils.data import DataLoader, TensorDataset\n\nimport os\n\nimport pandas as pd\n\nfrom sklearn.preprocessing import StandardScaler\n\nfrom tqdm import tqdm\n\nfrom concurrent.futures import ThreadPoolExecutor\n\n\n\n# Check if GPU is available\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nprint(f\"Using device: {device}\")\n\n\n\n# Define base directories for train and test series files\n\ntrain_base_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/'\n\ntest_base_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/'\n\n\n\n# Process a single file\n\ndef process_file(filename, dirname):\n\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n\n    df.drop('step', axis=1, inplace=True)\n\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\n\n# Load time series data using multithreading\n\ndef load_time_series(dirname) -> pd.DataFrame:\n\n    ids = os.listdir(dirname)\n\n    \n\n    with ThreadPoolExecutor() as executor:\n\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    \n\n    stats, indexes = zip(*results)\n\n    \n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n\n    df['id'] = indexes\n\n    return df\n\n\n\n# Define the Autoencoder model\n\nclass ActigraphyAutoencoder(nn.Module):\n\n    def __init__(self, input_dim, encoding_dim=32):\n\n        super(ActigraphyAutoencoder, self).__init__()\n\n        self.encoder = nn.Sequential(\n\n            nn.Linear(input_dim, 64),\n\n            nn.ReLU(),\n\n            nn.Linear(64, encoding_dim),\n\n            nn.ReLU()\n\n        )\n\n        self.decoder = nn.Sequential(\n\n            nn.Linear(encoding_dim, 64),\n\n            nn.ReLU(),\n\n            nn.Linear(64, input_dim),\n\n            nn.Sigmoid()\n\n        )\n\n\n\n    def forward(self, x):\n\n        encoded = self.encoder(x)\n\n        decoded = self.decoder(encoded)\n\n        return encoded, decoded\n\n\n\n# Train the autoencoder\n\ndef train_autoencoder(model, train_data, num_epochs=200, batch_size=512, lr=0.001):\n\n    criterion = nn.MSELoss()\n\n    optimizer = torch.optim.Adam(model.parameters(), lr=lr)\n\n    train_loader = DataLoader(train_data, batch_size=batch_size, shuffle=True)\n\n\n\n    for epoch in range(num_epochs):\n\n        total_loss = 0\n\n        for batch in train_loader:\n\n            batch = batch.to(device)\n\n            optimizer.zero_grad()\n\n            _, decoded = model(batch)\n\n            loss = criterion(decoded, batch)\n\n            loss.backward()\n\n            optimizer.step()\n\n            total_loss += loss.item()\n\n        #print(f\"Epoch [{epoch+1}/{num_epochs}], Loss: {total_loss:.4f}\")\n\n\n\n# Encode data with the trained autoencoder\n\ndef encode_data(data, model):\n\n    with torch.no_grad():\n\n        encoded_data = model.encoder(data.to(device))\n\n        return pd.DataFrame(encoded_data.cpu().numpy(), columns=[f'encoded_{i}' for i in range(encoded_data.size(1))])\n\n\n\n# Load and scale train and test data\n\nscaler = StandardScaler()\n\ntrain_df = load_time_series(train_base_dir)\n\ntest_df = load_time_series(test_base_dir)\n\n\n\n# Fit the scaler on the train data and transform both train and test data\n\nscaled_train_data = scaler.fit_transform(train_df.drop('id', axis=1))\n\nscaled_test_data = scaler.transform(test_df.drop('id', axis=1))\n\n\n\n# Prepare input data for autoencoder training\n\ninput_dim = scaled_train_data.shape[1]\n\ntrain_data_tensor = torch.tensor(scaled_train_data, dtype=torch.float32)\n\n\n\n# Initialize the model\n\nencoding_dim = 32\n\nmodel = ActigraphyAutoencoder(input_dim=input_dim, encoding_dim=encoding_dim).to(device)\n\n\n\n# Train the autoencoder on the entire scaled dataset for 30 epochs\n\ntrain_autoencoder(model, train_data_tensor, num_epochs=300)\n\n\n\n# Encode train and test data\n\nencoded_train_data = encode_data(train_data_tensor, model)\n\nencoded_train_data['id'] = train_df['id']\n\nencoded_test_data = encode_data(torch.tensor(scaled_test_data, dtype=torch.float32), model)\n\nencoded_test_data['id'] = test_df['id']\n\n\n\n# Save merged train and test DataFrames as CSV files in Kaggle working directory\n\nencoded_train_data.to_csv('/kaggle/working/encoded_train_data.csv', index=False)\n\nencoded_test_data.to_csv('/kaggle/working/encoded_test_data.csv', index=False)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:24:44.881621Z","iopub.execute_input":"2024-11-20T10:24:44.882146Z","iopub.status.idle":"2024-11-20T10:26:21.269770Z","shell.execute_reply.started":"2024-11-20T10:24:44.882097Z","shell.execute_reply":"2024-11-20T10:26:21.268314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the encoded train and test data\n\nencoded_train_df = pd.read_csv('/kaggle/working/encoded_train_data.csv')\n\nencoded_test_df = pd.read_csv('/kaggle/working/encoded_test_data.csv')\n\n\n\n# Load the main train and test datasets\n\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n\n\n# Merge the main train dataset with the encoded train data\n\nmerged_train_df = train_df.merge(encoded_train_df, on='id', how='left')\n\n\n\n# Merge the main test dataset with the encoded test data\n\nmerged_test_df = test_df.merge(encoded_test_df, on='id', how='left')\n\n\n\n# Save the merged datasets as new CSV files\n\nmerged_train_df.to_csv('/kaggle/working/final_merged_train_data.csv', index=False)\n\nmerged_test_df.to_csv('/kaggle/working/final_merged_test_data.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.271868Z","iopub.execute_input":"2024-11-20T10:26:21.272215Z","iopub.status.idle":"2024-11-20T10:26:21.628512Z","shell.execute_reply.started":"2024-11-20T10:26:21.272182Z","shell.execute_reply":"2024-11-20T10:26:21.627033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_train_df =pd.read_csv('/kaggle/working/final_merged_train_data.csv')\n\nmerged_test_df = pd.read_csv('/kaggle/working/final_merged_test_data.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.630294Z","iopub.execute_input":"2024-11-20T10:26:21.630809Z","iopub.status.idle":"2024-11-20T10:26:21.691594Z","shell.execute_reply.started":"2024-11-20T10:26:21.630738Z","shell.execute_reply":"2024-11-20T10:26:21.690515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify columns that are in the train data but not in the test data, except 'sii' (target variable)\ncolumns_to_drop = [col for col in merged_train_df.columns if col not in merged_test_df.columns and col != 'sii']\n\n# Drop the identified columns from the training data\nmerged_train_df = merged_train_df.drop(columns=columns_to_drop)\n\n# Confirm the new shape of the training data\nprint(\"After dropping columns not in test data (except 'sii'):\")\nprint(\"Train Data Shape:\", merged_train_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.692743Z","iopub.execute_input":"2024-11-20T10:26:21.693098Z","iopub.status.idle":"2024-11-20T10:26:21.705117Z","shell.execute_reply.started":"2024-11-20T10:26:21.693065Z","shell.execute_reply":"2024-11-20T10:26:21.703744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove rows with missing 'sii' values in the train data\nmerged_train_df = merged_train_df.dropna(subset=['sii'])\n\n# Confirm that rows with missing 'sii' values have been removed\nprint(\"After removing missing 'sii' rows:\")\nprint(\"Train Data Shape:\", merged_train_df.shape)\nprint(\"\\nMissing Values in 'sii':\", merged_train_df['sii'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.709025Z","iopub.execute_input":"2024-11-20T10:26:21.709408Z","iopub.status.idle":"2024-11-20T10:26:21.722858Z","shell.execute_reply.started":"2024-11-20T10:26:21.709358Z","shell.execute_reply":"2024-11-20T10:26:21.721609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.724496Z","iopub.execute_input":"2024-11-20T10:26:21.725099Z","iopub.status.idle":"2024-11-20T10:26:21.764679Z","shell.execute_reply.started":"2024-11-20T10:26:21.725048Z","shell.execute_reply":"2024-11-20T10:26:21.763444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the season mapping\nseason_mapping = {\n    'Spring': 1,\n    'Summer': 2,\n    'Fall': 3,\n    'Winter': 4\n}\n\n# List of columns with season data\nseason_columns = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n    'PAQ_A-Season', 'PAQ_C-Season', \n    'SDS-Season', 'PreInt_EduHx-Season'\n]\n\n# Apply the mapping to each season column in both train and test dataframes\nfor col in season_columns:\n    merged_train_df[col] = merged_train_df[col].map(season_mapping)\n    merged_test_df[col] = merged_test_df[col].map(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.766221Z","iopub.execute_input":"2024-11-20T10:26:21.766668Z","iopub.status.idle":"2024-11-20T10:26:21.793509Z","shell.execute_reply.started":"2024-11-20T10:26:21.766621Z","shell.execute_reply":"2024-11-20T10:26:21.792602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering function\ndef feature_engineering(df):\n    # Your feature engineering code here\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    # Interaction features\n    #df['Height_Age_Interaction'] = df['Physical-Height'] * df['Basic_Demos-Age']\n    #df['Weight_Height_Interaction'] = df['Physical-Weight'] * df['Physical-Height']\n\n    # Weight-Height Ratio\n    #df['Weight_Height_Ratio'] = df['Physical-Weight'] / df['Physical-Height']\n\n    # Age Binning\n    unique_ages = df['Basic_Demos-Age'].unique()\n    min_age = int(min(unique_ages))\n    max_age = int(max(unique_ages))\n    age_range = max_age - min_age\n    bin_size = max(1, age_range // 5 + 1)\n    age_bins = list(range(min_age, max_age + bin_size, bin_size))\n    age_labels = [f'{age}-{age + bin_size - 1}' for age in age_bins[:-1]]\n    #df['Age_Bin'] = pd.cut(df['Basic_Demos-Age'], bins=age_bins, labels=age_labels, right=False)\n    age_bin_mapping = {label: idx + 1 for idx, label in enumerate(age_labels)}\n    #df['Age_Bin'] = df['Age_Bin'].map(age_bin_mapping).astype('Int64')\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.795064Z","iopub.execute_input":"2024-11-20T10:26:21.795490Z","iopub.status.idle":"2024-11-20T10:26:21.807553Z","shell.execute_reply.started":"2024-11-20T10:26:21.795443Z","shell.execute_reply":"2024-11-20T10:26:21.806214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.809078Z","iopub.execute_input":"2024-11-20T10:26:21.809534Z","iopub.status.idle":"2024-11-20T10:26:21.846261Z","shell.execute_reply.started":"2024-11-20T10:26:21.809474Z","shell.execute_reply":"2024-11-20T10:26:21.845058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_train_df=merged_train_df\nfiltered_test_df=merged_test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.847685Z","iopub.execute_input":"2024-11-20T10:26:21.848090Z","iopub.status.idle":"2024-11-20T10:26:21.862532Z","shell.execute_reply.started":"2024-11-20T10:26:21.848056Z","shell.execute_reply":"2024-11-20T10:26:21.861359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Engineering\nfiltered_train_df = feature_engineering(filtered_train_df)\nfiltered_test_df = feature_engineering(filtered_test_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.864347Z","iopub.execute_input":"2024-11-20T10:26:21.864815Z","iopub.status.idle":"2024-11-20T10:26:21.893755Z","shell.execute_reply.started":"2024-11-20T10:26:21.864740Z","shell.execute_reply":"2024-11-20T10:26:21.892638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.895220Z","iopub.execute_input":"2024-11-20T10:26:21.895649Z","iopub.status.idle":"2024-11-20T10:26:21.924294Z","shell.execute_reply.started":"2024-11-20T10:26:21.895602Z","shell.execute_reply":"2024-11-20T10:26:21.923029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.925716Z","iopub.execute_input":"2024-11-20T10:26:21.926125Z","iopub.status.idle":"2024-11-20T10:26:21.953155Z","shell.execute_reply.started":"2024-11-20T10:26:21.926077Z","shell.execute_reply":"2024-11-20T10:26:21.951933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_train_df['sii'] = filtered_train_df['sii'].round().astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.956861Z","iopub.execute_input":"2024-11-20T10:26:21.957198Z","iopub.status.idle":"2024-11-20T10:26:21.963145Z","shell.execute_reply.started":"2024-11-20T10:26:21.957169Z","shell.execute_reply":"2024-11-20T10:26:21.961992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(filtered_train_df['sii'].dtype)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.964557Z","iopub.execute_input":"2024-11-20T10:26:21.964941Z","iopub.status.idle":"2024-11-20T10:26:21.978969Z","shell.execute_reply.started":"2024-11-20T10:26:21.964907Z","shell.execute_reply":"2024-11-20T10:26:21.977731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import classification_report, cohen_kappa_score\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.ensemble import VotingClassifier\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom scipy.optimize import minimize\n\n# Threshold rounding and evaluation functions\ndef quadratic_weighted_kappa(y_true, y_pred):\n    \"\"\"Calculate Quadratic Weighted Kappa (QWK) score.\"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(predictions, thresholds):\n    \"\"\"Round the predictions based on optimized thresholds.\"\"\"\n    return np.where(predictions < thresholds[0], 0,\n                    np.where(predictions < thresholds[1], 1,\n                             np.where(predictions < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, predictions):\n    \"\"\"Evaluate predictions with QWK based on thresholding.\"\"\"\n    rounded_predictions = threshold_Rounder(predictions, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_predictions)\n\n# K-fold cross-validation setup\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Models and parameters\nlgb_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_child_samples': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'reg_alpha': 10,\n    'reg_lambda': 0.01,\n    'random_state': 42\n}\n\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': 42,\n    'use_label_encoder': False,\n    'eval_metric': 'logloss'\n}\n\ncat_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,\n    'silent': True,\n    'l2_leaf_reg': 10\n}\n\n# Initialize models\nxgb_model = XGBClassifier(**xgb_params, verbosity=0)  # Suppress output\nlgb_model = LGBMClassifier(**lgb_params, n_estimators=300, verbose=-1)  # Suppress output\ncat_model = CatBoostClassifier(**cat_params)  # Suppress output\n\n# Create a Voting Classifier\nvoting_clf1 = VotingClassifier(\n    estimators=[('xgb', xgb_model), ('lgb', lgb_model), ('cat', cat_model)],\n    voting='soft'\n)\n\n# Initialize variables to store results\nall_qwk_scores = []\noptimized_thresholds_list = []\ntest_preds = np.zeros((filtered_test_df.shape[0], 4))  # Initialize for fold test predictions\n\n# Separate the 'id' column from both filtered train and test DataFrames\ntrain_ids = filtered_train_df['id'].values\ntest_ids = filtered_test_df['id'].values\n\n# Drop 'id' and 'sii' columns from training data\nfiltered_train_df_no_id_no_sii = filtered_train_df.drop(columns=['id', 'sii'])\nsii_column = filtered_train_df['sii'].values\n\noof_predictions = np.zeros_like(sii_column, dtype=float)  # For storing OOF continuous predictions\noof_true_labels = np.zeros_like(sii_column, dtype=int)  # For storing OOF true labels\ntest_preds = np.zeros((filtered_test_df.shape[0], 4))  # Initialize for fold test predictions\n\n# Start K-fold cross-validation\nfor train_index, val_index in kf.split(filtered_train_df_no_id_no_sii, sii_column):\n    # Split the data into training and validation sets\n    train_data, val_data = filtered_train_df_no_id_no_sii.iloc[train_index], filtered_train_df_no_id_no_sii.iloc[val_index]\n    sii_train, sii_val = sii_column[train_index], sii_column[val_index]\n\n    # Apply random oversampling\n    oversampler = RandomOverSampler(random_state=42)\n    X_train, y_train = oversampler.fit_resample(train_data, sii_train)\n    X_val = val_data\n    y_val = sii_val\n\n    \n    #X_train = train_data\n    #y_train = sii_train\n    #X_val = val_data\n    #y_val = sii_val\n\n    # Fit the Voting Classifier\n    voting_clf1.fit(X_train, y_train)\n\n    # Predict probabilities for validation and test data\n    y_val_pred_proba = voting_clf1.predict_proba(X_val)\n    test_fold_preds = voting_clf1.predict_proba(filtered_test_df.drop(columns=['id']))\n    test_preds += test_fold_preds  # Accumulate test predictions across folds\n\n    # Continuous predictions and store them for OOF optimization\n    y_val_pred_continuous = np.dot(y_val_pred_proba, np.arange(y_val_pred_proba.shape[1]))\n    oof_predictions[val_index] = y_val_pred_continuous  # Store OOF predictions\n    oof_true_labels[val_index] = y_val  # Store OOF true labels\n\n    # Perform fold-wise threshold optimization\n    KappaOptimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5],\n        args=(y_val, y_val_pred_continuous),\n        method='Nelder-Mead'\n    )\n    optimized_thresholds = KappaOptimizer.x\n    optimized_thresholds_list.append(optimized_thresholds)\n\n    # Apply thresholds and calculate QWK score\n    y_val_pred_tuned = threshold_Rounder(y_val_pred_continuous, optimized_thresholds)\n    qwk_score = quadratic_weighted_kappa(y_val, y_val_pred_tuned)\n    all_qwk_scores.append(qwk_score)\n\n# Calculate the mean QWK score across all folds\nmean_qwk_score = np.mean(all_qwk_scores)\nprint(f\"Mean QWK Score across all folds: {mean_qwk_score:.4f}\")\n\n# Final threshold optimization on OOF predictions\nfinal_optimizer = minimize(\n    evaluate_predictions,\n    x0=[0.5, 1.5, 2.5],\n    args=(oof_true_labels, oof_predictions),\n    method='Nelder-Mead'\n)\nfinal_optimized_thresholds = final_optimizer.x\nprint(f\"Final Optimized Thresholds: {final_optimized_thresholds}\")\n\n# Average test predictions across all folds\ntest_preds_mean = test_preds / kf.n_splits\n\n# Generate final test predictions\ncontinuous_test_predictions = np.dot(test_preds_mean, np.arange(test_preds_mean.shape[1]))\nfinal_predictions_test = threshold_Rounder(continuous_test_predictions, final_optimized_thresholds)\n\n# Create a submission DataFrame\nsubmission_df = pd.DataFrame({\n    'id': test_ids,\n    'sii': final_predictions_test\n})\n\n# Save submission file\n#submission_df.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:26:21.980708Z","iopub.execute_input":"2024-11-20T10:26:21.981105Z","iopub.status.idle":"2024-11-20T10:28:22.290911Z","shell.execute_reply.started":"2024-11-20T10:26:21.981072Z","shell.execute_reply":"2024-11-20T10:28:22.289735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:28:22.292415Z","iopub.execute_input":"2024-11-20T10:28:22.292842Z","iopub.status.idle":"2024-11-20T10:28:22.303656Z","shell.execute_reply.started":"2024-11-20T10:28:22.292794Z","shell.execute_reply":"2024-11-20T10:28:22.302492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.ensemble import VotingClassifier\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom scipy.optimize import differential_evolution\n\n# Threshold rounding and evaluation functions\ndef quadratic_weighted_kappa(y_true, y_pred):\n    \"\"\"Calculate Quadratic Weighted Kappa (QWK) score.\"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(predictions, thresholds):\n    \"\"\"Round the predictions based on optimized thresholds.\"\"\"\n    return np.where(predictions < thresholds[0], 0,\n                    np.where(predictions < thresholds[1], 1,\n                             np.where(predictions < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, predictions):\n    \"\"\"Evaluate predictions with QWK based on thresholding.\"\"\"\n    rounded_predictions = threshold_Rounder(predictions, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_predictions)\n\nfrom scipy.optimize import minimize\n\ndef optimize_thresholds_nelder_mead(y_true, predictions):\n    \"\"\"\n    Optimize thresholds using Nelder-Mead for maximizing Quadratic Weighted Kappa (QWK).\n    \n    Parameters:\n    - y_true: Ground truth labels.\n    - predictions: Continuous predictions from the model.\n\n    Returns:\n    - Optimized thresholds (array).\n    - Maximum QWK score (float).\n    \"\"\"\n    # Define the initial thresholds\n    initial_thresholds = [0.5, 1.5, 2.5]\n    \n    # Use Nelder-Mead optimization with default settings\n    result = minimize(\n        evaluate_predictions,  # Function to minimize (negative QWK)\n        x0=initial_thresholds,  # Initial guess for thresholds\n        args=(y_true, predictions),  # Additional arguments for the evaluation function\n        method='Nelder-Mead',  # Optimization method\n    )\n    \n    return result.x, -result.fun\n\n\n# K-fold cross-validation setup\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Models and parameters\nlgb_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_child_samples': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'reg_alpha': 20,\n    'reg_lambda': 0.05,\n    'random_state': 42\n}\n\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1.5,\n    'reg_lambda': 10,\n    'random_state': 42,\n    'use_label_encoder': False,\n    'eval_metric': 'logloss'\n}\n\ncat_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,\n    'silent': True,\n    'l2_leaf_reg': 15\n}\n\n# Initialize models\nxgb_model = XGBClassifier(**xgb_params, verbosity=0)  # Suppress output\nlgb_model = LGBMClassifier(**lgb_params, n_estimators=300, verbose=-1)  # Suppress output\ncat_model = CatBoostClassifier(**cat_params)  # Suppress output\n\n# Create a Voting Classifier\nvoting_clf1 = VotingClassifier(\n    estimators=[('xgb', xgb_model), ('lgb', lgb_model), ('cat', cat_model)],\n    voting='soft'\n)\n\n# Initialize variables to store results\noof_predictions = np.zeros(len(filtered_train_df))  # To store continuous predictions\noof_true_labels = np.zeros(len(filtered_train_df))  # To store true labels\ntest_preds = np.zeros((filtered_test_df.shape[0], 4))  # Initialize for fold test predictions\n\n# Drop 'id' and 'sii' columns from training data\nfiltered_train_df_no_id_no_sii = filtered_train_df.drop(columns=['id', 'sii'])\nsii_column = filtered_train_df['sii'].values\n\n# Start K-fold cross-validation\nfor train_index, val_index in kf.split(filtered_train_df_no_id_no_sii, sii_column):\n    # Split the data into training and validation sets\n    train_data, val_data = filtered_train_df_no_id_no_sii.iloc[train_index], filtered_train_df_no_id_no_sii.iloc[val_index]\n    sii_train, sii_val = sii_column[train_index], sii_column[val_index]\n\n    # Apply random oversampling\n   # oversampler = RandomOverSampler(random_state=42)\n    #X_train, y_train = oversampler.fit_resample(train_data, sii_train)\n    #X_val = val_data\n    #y_val = sii_val\n\n    X_train = train_data\n    y_train = sii_train\n    X_val = val_data\n    y_val = sii_val\n\n    # Fit the Voting Classifier\n    voting_clf1.fit(X_train, y_train)\n\n    # Predict probabilities for validation and test data\n    y_val_pred_proba = voting_clf1.predict_proba(X_val)\n    test_fold_preds = voting_clf1.predict_proba(filtered_test_df.drop(columns=['id']))\n    test_preds += test_fold_preds  # Accumulate test predictions across folds\n\n    # Compute continuous predictions for validation data\n    y_val_pred_continuous = np.dot(y_val_pred_proba, np.arange(y_val_pred_proba.shape[1]))\n\n    # Store out-of-fold predictions and true labels\n    oof_predictions[val_index] = y_val_pred_continuous\n    oof_true_labels[val_index] = y_val\n\n# Optimize thresholds on the combined out-of-fold predictions\n# Optimize thresholds on the combined out-of-fold predictions using Nelder-Mead\nfinal_optimized_thresholds, _ = optimize_thresholds_nelder_mead(oof_true_labels, oof_predictions)\n\n\n# Apply the optimized thresholds to round the OOF predictions\nrounded_oof_predictions = threshold_Rounder(oof_predictions, final_optimized_thresholds)\n\n# Calculate the final QWK score\nfinal_qwk_score = quadratic_weighted_kappa(oof_true_labels, rounded_oof_predictions)\n\n# Print results\nprint(f\"Optimized Thresholds: {final_optimized_thresholds}\")\nprint(f\"Final QWK Score (Validation): {final_qwk_score:.4f}\")\n\n# Average test predictions across all folds\ntest_preds_mean = test_preds / kf.n_splits\n\n# Generate final test predictions using the global optimized thresholds\ncontinuous_test_predictions = np.dot(test_preds_mean, np.arange(test_preds_mean.shape[1]))\nfinal_predictions_test = threshold_Rounder(continuous_test_predictions, final_optimized_thresholds)\n\n# Create a submission DataFrame\nsubmission_df2 = pd.DataFrame({\n    'id': filtered_test_df['id'],\n    'sii': final_predictions_test\n})\n\n# Save submission to CSV\n#submission_df2.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:28:22.305136Z","iopub.execute_input":"2024-11-20T10:28:22.305477Z","iopub.status.idle":"2024-11-20T10:29:53.373951Z","shell.execute_reply.started":"2024-11-20T10:28:22.305443Z","shell.execute_reply":"2024-11-20T10:29:53.372758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df2.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:29:53.375486Z","iopub.execute_input":"2024-11-20T10:29:53.375987Z","iopub.status.idle":"2024-11-20T10:29:53.387847Z","shell.execute_reply.started":"2024-11-20T10:29:53.375933Z","shell.execute_reply":"2024-11-20T10:29:53.386419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T11:05:10.317029Z","iopub.execute_input":"2024-11-20T11:05:10.317590Z","iopub.status.idle":"2024-11-20T11:05:15.093912Z","shell.execute_reply.started":"2024-11-20T11:05:10.317540Z","shell.execute_reply":"2024-11-20T11:05:15.092447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T11:05:19.739702Z","iopub.execute_input":"2024-11-20T11:05:19.740388Z","iopub.status.idle":"2024-11-20T11:06:59.844137Z","shell.execute_reply.started":"2024-11-20T11:05:19.740343Z","shell.execute_reply":"2024-11-20T11:06:59.840053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.optimize import differential_evolution\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Differential Evolution for threshold optimization\n    bounds = [(0, 0.7), (0.7, 1.7), (1.7, 2.7)] # Ensure valid thresholds\n    result = differential_evolution(evaluate_predictions, bounds, args=(y, oof_non_rounded), strategy='best1bin', \n                                    maxiter=100, popsize=15, tol=1e-7, seed=SEED)\n    assert result.success, \"Optimization did not converge.\"\n    \n    optimized_thresholds = result.x\n    oof_tuned = threshold_Rounder(oof_non_rounded, optimized_thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, optimized_thresholds)\n    \n    submission3 = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission3, model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T11:25:48.679941Z","iopub.execute_input":"2024-11-20T11:25:48.680386Z","iopub.status.idle":"2024-11-20T11:25:48.696935Z","shell.execute_reply.started":"2024-11-20T11:25:48.680347Z","shell.execute_reply":"2024-11-20T11:25:48.695729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nParams7 = {'learning_rate': 0.03884249148676395, 'max_depth': 12, 'num_leaves': 413, 'min_data_in_leaf': 14,\n           'feature_fraction': 0.7987976913702801, 'bagging_fraction': 0.7602261703576205, 'bagging_freq': 2, \n           'lambda_l1': 4.735462555910575, 'lambda_l2': 4.735028557007343e-06} \n\nLight = lgb.LGBMRegressor(**Params7,random_state=SEED, verbose=-1,n_estimators=200)\nSubmission3,model = TrainML(Light,test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T11:25:56.142421Z","iopub.execute_input":"2024-11-20T11:25:56.142842Z","iopub.status.idle":"2024-11-20T11:26:20.258442Z","shell.execute_reply.started":"2024-11-20T11:25:56.142804Z","shell.execute_reply":"2024-11-20T11:26:20.257317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = submission_df\n\n\n\nsub2 = submission_df2\n\n\n\nsub3 = Submission3\n\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\n\n\n\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\n\n\n\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\n\n\n\n\n\n\ncombined = pd.DataFrame({\n\n\n\n    'id': sub1['id'],\n\n\n\n    'sii_1': sub1['sii'],\n\n\n\n    'sii_2': sub2['sii'],\n\n\n\n    'sii_3': sub3['sii']\n\n\n\n})\n\n\n\n\n\n\n\ndef majority_vote(row):\n\n\n\n    return row.mode()[0]\n\n\n\n\n\n\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\n\n\n\n\n\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n\n\n\n\n\n\nfinal_submission.to_csv('submission.csv', index=False)\n\n\n\n\n\n\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:31:38.321499Z","iopub.execute_input":"2024-11-20T10:31:38.321829Z","iopub.status.idle":"2024-11-20T10:31:38.345152Z","shell.execute_reply.started":"2024-11-20T10:31:38.321774Z","shell.execute_reply":"2024-11-20T10:31:38.343815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T10:31:38.346533Z","iopub.execute_input":"2024-11-20T10:31:38.346951Z","iopub.status.idle":"2024-11-20T10:31:38.365530Z","shell.execute_reply.started":"2024-11-20T10:31:38.346916Z","shell.execute_reply":"2024-11-20T10:31:38.364240Z"}},"outputs":[],"execution_count":null}]}