{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import VotingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n\nSEED = 42\nN_SPLITS = 5\nENCODING_DIM = 96","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:49:23.192001Z","iopub.execute_input":"2024-11-02T17:49:23.193014Z","iopub.status.idle":"2024-11-02T17:49:31.815061Z","shell.execute_reply.started":"2024-11-02T17:49:23.192901Z","shell.execute_reply":"2024-11-02T17:49:31.813425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set feature and categorical columns\nFEATURE_COLS = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', \n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', \n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', \n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', \n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', \n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', \n                'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'SDS-Season', \n                'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nCATEGORICAL_COLS = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n                    'PAQ_A-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:49:32.164422Z","iopub.execute_input":"2024-11-02T17:49:32.165978Z","iopub.status.idle":"2024-11-02T17:49:32.173438Z","shell.execute_reply.started":"2024-11-02T17:49:32.165922Z","shell.execute_reply":"2024-11-02T17:49:32.171821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------\n# 1. LOAD AND PREPROCESS TIME SERIES DATA\n# -------------------------\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    stats, indexes = [], []\n\n    for fname in tqdm(ids, desc=\"Loading time series\"):\n        stat, idx = process_file(fname, dirname)\n        stats.append(stat)\n        indexes.append(idx)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:49:42.962288Z","iopub.execute_input":"2024-11-02T17:49:42.962749Z","iopub.status.idle":"2024-11-02T17:49:42.972894Z","shell.execute_reply.started":"2024-11-02T17:49:42.962704Z","shell.execute_reply":"2024-11-02T17:49:42.971738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# -------------------------\n# 2. BUILD AND TRAIN AUTOENCODER FOR DIMENSIONALITY REDUCTION\n# -------------------------\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(nn.Linear(input_dim, encoding_dim), nn.ReLU())\n        self.decoder = nn.Sequential(nn.Linear(encoding_dim, input_dim), nn.Sigmoid())\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\ndef perform_autoencoder_torch(df, encoding_dim=ENCODING_DIM, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n    return pd.DataFrame(encoded_data, columns=[f'Enc_{i+1}' for i in range(encoded_data.shape[1])])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:49:54.832255Z","iopub.execute_input":"2024-11-02T17:49:54.832755Z","iopub.status.idle":"2024-11-02T17:49:54.849251Z","shell.execute_reply.started":"2024-11-02T17:49:54.832709Z","shell.execute_reply":"2024-11-02T17:49:54.847932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------\n# 3. LOAD MAIN DATA AND MERGE ENCODED TIME SERIES FEATURES\n# -------------------------\n# Load CSV data\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Load and encode time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:57:16.102001Z","iopub.execute_input":"2024-11-02T17:57:16.102583Z","iopub.status.idle":"2024-11-02T18:00:04.990312Z","shell.execute_reply.started":"2024-11-02T17:57:16.102525Z","shell.execute_reply":"2024-11-02T18:00:04.988742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_encoded = perform_autoencoder_torch(train_ts.drop('id', axis=1))\ntrain_ts_encoded['id'] = train_ts['id']\n\n# Merge encoded features back to main data\ntrain = pd.merge(train_df, train_ts_encoded, on='id', how='left')\ntest = pd.merge(test_df, train_ts_encoded, on='id', how='left')\n\n# Select features and target\ntrain = train[FEATURE_COLS].dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T18:00:04.992820Z","iopub.execute_input":"2024-11-02T18:00:04.993269Z","iopub.status.idle":"2024-11-02T18:00:07.566096Z","shell.execute_reply.started":"2024-11-02T18:00:04.993220Z","shell.execute_reply":"2024-11-02T18:00:07.564338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------\n# 4. PROCESS CATEGORICAL FEATURES\n# -------------------------\ndef update_categorical(df):\n    for col in CATEGORICAL_COLS:\n        df[col] = df[col].fillna('Missing').astype('category')\n    return df\n\ntrain = update_categorical(train)\ntest = update_categorical(test)\n\n# Encode categorical features to integers\nfor col in CATEGORICAL_COLS:\n    mapping = {cat: idx for idx, cat in enumerate(train[col].cat.categories)}\n    train[col] = train[col].map(mapping).astype(int)\n    test[col] = test[col].map(mapping).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T18:00:07.638290Z","iopub.execute_input":"2024-11-02T18:00:07.638881Z","iopub.status.idle":"2024-11-02T18:00:07.685978Z","shell.execute_reply.started":"2024-11-02T18:00:07.638822Z","shell.execute_reply":"2024-11-02T18:00:07.683589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_columns = [col for col in train.columns if col != 'sii']\ntrain = train[feature_columns + ['sii']]\ntest = test[feature_columns]\n\n# Verify shapes to confirm they match\nprint(\"Train shape (aligned):\", train.shape)\nprint(\"Test shape (aligned):\", test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T18:00:07.740925Z","iopub.execute_input":"2024-11-02T18:00:07.741532Z","iopub.status.idle":"2024-11-02T18:00:07.759945Z","shell.execute_reply.started":"2024-11-02T18:00:07.741452Z","shell.execute_reply":"2024-11-02T18:00:07.758296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------\n# 5. TRAIN MODEL WITH STRATIFIED K-FOLD CROSS-VALIDATION\n# -------------------------\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_rounder(preds, thresholds):\n    return np.where(preds < thresholds[0], 0, \n                    np.where(preds < thresholds[1], 1, \n                             np.where(preds < thresholds[2], 2, 3)))\n\ndef optimize_thresholds(y_true, preds):\n    return minimize(lambda th: -quadratic_weighted_kappa(y_true, threshold_rounder(preds, th)), \n                    x0=[0.5, 1.5, 2.5], method='Nelder-Mead').x\n\ndef train_model(model, train, test):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    test_preds = np.zeros((len(test), N_SPLITS))\n    oof_preds = np.zeros(len(y))\n\n    skf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)\n    for fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        model.fit(X_train, y_train)\n        oof_preds[val_idx] = model.predict(X_val)\n        test_preds[:, fold] = model.predict(test)\n\n    optimized_thresholds = optimize_thresholds(y, oof_preds)\n    oof_preds_rounded = threshold_rounder(oof_preds, optimized_thresholds)\n\n    print(f\"Optimized QWK: {quadratic_weighted_kappa(y, oof_preds_rounded):.4f}\")\n    \n    final_test_preds = threshold_rounder(test_preds.mean(axis=1), optimized_thresholds)\n    return pd.DataFrame({'id': sample_submission['id'], 'sii': final_test_preds})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T17:53:38.959776Z","iopub.execute_input":"2024-11-02T17:53:38.960292Z","iopub.status.idle":"2024-11-02T17:53:38.976792Z","shell.execute_reply.started":"2024-11-02T17:53:38.960223Z","shell.execute_reply":"2024-11-02T17:53:38.975034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------\n# 6. DEFINE AND TRAIN VOTING REGRESSOR\n# -------------------------\n\n\n# Define model-specific parameters\nlightgbm_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'random_state': SEED\n}\n\nxgboost_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'random_state': SEED\n}\n\ncatboost_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0\n}\n\n# Initialize models with their respective parameters\nLight = LGBMRegressor(**lightgbm_params)\nXGB_Model = XGBRegressor(**xgboost_params)\nCatBoost_Model = CatBoostRegressor(**catboost_params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n\nvoting_model = VotingRegressor(estimators=[\n    ('lgbm', Light), ('xgb', XGB_Model), ('catboost', CatBoost_Model)\n])\n\nsubmission = train_model(voting_model, train, test)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T18:00:15.737168Z","iopub.execute_input":"2024-11-02T18:00:15.737629Z","iopub.status.idle":"2024-11-02T18:00:22.843206Z","shell.execute_reply.started":"2024-11-02T18:00:15.737584Z","shell.execute_reply":"2024-11-02T18:00:22.841983Z"}},"outputs":[],"execution_count":null}]}