{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport os\nfrom concurrent.futures import ThreadPoolExecutor\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-12T05:47:21.472937Z","iopub.execute_input":"2024-10-12T05:47:21.473330Z","iopub.status.idle":"2024-10-12T05:47:21.486447Z","shell.execute_reply.started":"2024-10-12T05:47:21.473295Z","shell.execute_reply":"2024-10-12T05:47:21.485635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom imblearn.over_sampling import SMOTE\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Load time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T05:50:07.726044Z","iopub.execute_input":"2024-10-12T05:50:07.726759Z","iopub.status.idle":"2024-10-12T05:51:33.670528Z","shell.execute_reply.started":"2024-10-12T05:50:07.726719Z","shell.execute_reply":"2024-10-12T05:51:33.669569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge datasets\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n# Define feature columns\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score',\n                'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n                'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n                'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',\n                'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', 'sii']\nfeaturesCols += time_series_cols\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season',\n         'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c:\n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(create_mapping(col, test)).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1).values\n    y = train['sii'].values\n\n    imputer = SimpleImputer(strategy='mean')\n    X = imputer.fit_transform(X)\n    test_data = imputer.transform(test_data)\n\n    #smote = SMOTE(random_state=SEED)\n    #X_res, y_res = smote.fit_resample(X, y)\n    \n    X_res, y_res = X, y\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    train_S = []\n    test_S = []\n    oof_non_rounded = np.zeros(len(y_res), dtype=float)\n    oof_rounded = np.zeros(len(y_res), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, val_idx) in enumerate(SKF.split(X_res, y_res)):\n        X_train, X_val = X_res[train_idx], X_res[val_idx]\n        y_train, y_val = y_res[train_idx], y_res[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[val_idx] = y_val_pred\n        y_val_pred_rounded = np.round(y_val_pred).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, np.round(y_train_pred).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimize thresholds with Nelder-Mead method\n    KappaOptimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y_res, oof_non_rounded), method='Nelder-Mead')\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y_res, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-10-12T05:54:45.969710Z","iopub.execute_input":"2024-10-12T05:54:45.970574Z","iopub.status.idle":"2024-10-12T05:54:45.984367Z","shell.execute_reply.started":"2024-10-12T05:54:45.970531Z","shell.execute_reply":"2024-10-12T05:54:45.983374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom scipy.optimize import minimize\nfrom IPython.display import clear_output\n\ndef build_nn_model(input_shape):\n    model = Sequential()\n    model.add(Dense(128, input_dim=input_shape, activation='relu'))\n    model.add(Dense(64, activation='relu'))\n    model.add(Dense(32, activation='relu'))\n    model.add(Dense(1, activation='linear'))  # Linear activation for regression\n\n    model.compile(optimizer=Adam(learning_rate=0.001), loss='mse')\n    return model\n\ndef TrainML( test_data):\n    # Prepare features and target\n    X = train.drop(['sii'], axis=1).values\n    y = train['sii'].values\n\n    # Handle missing values\n    imputer = SimpleImputer(strategy='mean')\n    X = imputer.fit_transform(X)\n    test_data = imputer.transform(test_data)\n\n    # Scale features\n    scaler = StandardScaler()\n    X = scaler.fit_transform(X)\n    test_data = scaler.transform(test_data)\n\n    # Resampling if needed\n    X_res, y_res = X, y  # Uncomment SMOTE if necessary\n\n    # Initialize stratified k-fold\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    train_S = []\n    test_S = []\n    oof_non_rounded = np.zeros(len(y_res), dtype=float)\n    oof_rounded = np.zeros(len(y_res), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, val_idx) in enumerate(SKF.split(X_res, y_res)):\n        X_train, X_val = X_res[train_idx], X_res[val_idx]\n        y_train, y_val = y_res[train_idx], y_res[val_idx]\n\n        # Build a new model for each fold\n        model = build_nn_model(X_train.shape[1])\n\n        # Early stopping\n        early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n        # Train the model\n        model.fit(X_train, y_train, epochs=100, batch_size=32, validation_data=(X_val, y_val),\n                  callbacks=[early_stopping], verbose=0)\n\n        # Predictions\n        y_train_pred = model.predict(X_train).flatten()\n        y_val_pred = model.predict(X_val).flatten()\n\n        # Store predictions for evaluation\n        oof_non_rounded[val_idx] = y_val_pred\n        y_val_pred_rounded = np.round(y_val_pred).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        # Calculate QWK scores\n        train_kappa = quadratic_weighted_kappa(y_train, np.round(y_train_pred).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        # Test predictions\n        test_preds[:, fold] = model.predict(test_data).flatten()\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimize thresholds\n    KappaOptimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y_res, oof_non_rounded), method='Nelder-Mead')\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y_res, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n\n    # Prepare submission\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:03:55.768734Z","iopub.execute_input":"2024-10-12T06:03:55.769595Z","iopub.status.idle":"2024-10-12T06:03:55.789104Z","shell.execute_reply.started":"2024-10-12T06:03:55.769553Z","shell.execute_reply":"2024-10-12T06:03:55.788082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n# Set random seeds for reproducibility\nSEED = 42\nnp.random.seed(SEED)\nrandom.seed(SEED)\n\n# Suppress warnings\nwarnings.filterwarnings('ignore')\n\n# Pandas option for displaying all columns\npd.options.display.max_columns = None\n# Constants\nn_splits = 5\n\n# Train the ensemble model\nSubmission = TrainML(test.values)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:03:55.946193Z","iopub.execute_input":"2024-10-12T06:03:55.946863Z","iopub.status.idle":"2024-10-12T06:04:22.797682Z","shell.execute_reply.started":"2024-10-12T06:03:55.946829Z","shell.execute_reply":"2024-10-12T06:04:22.796640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save submission\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-12T06:05:30.477269Z","iopub.execute_input":"2024-10-12T06:05:30.477937Z","iopub.status.idle":"2024-10-12T06:05:30.493648Z","shell.execute_reply.started":"2024-10-12T06:05:30.477897Z","shell.execute_reply":"2024-10-12T06:05:30.492814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}