{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport random\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport glob\nimport scipy\n\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.preprocessing import StandardScaler\nimport lightgbm as lgb\nimport optuna\nimport time\nfrom optuna.samplers import TPESampler\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom scipy.optimize import minimize\nfrom collections import Counter\nfrom scipy import stats\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay","metadata":{"execution":{"iopub.status.busy":"2024-12-20T10:24:01.838875Z","iopub.execute_input":"2024-12-20T10:24:01.841240Z","iopub.status.idle":"2024-12-20T10:24:01.857413Z","shell.execute_reply.started":"2024-12-20T10:24:01.841127Z","shell.execute_reply":"2024-12-20T10:24:01.856048Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:15:14.182370Z","iopub.execute_input":"2024-12-20T10:15:14.182935Z","iopub.status.idle":"2024-12-20T10:15:14.188181Z","shell.execute_reply.started":"2024-12-20T10:15:14.182896Z","shell.execute_reply":"2024-12-20T10:15:14.187010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_global_seed(seed=0):\n    np.random.seed(seed)\n    random.seed(seed)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def describe_x(df):\n    X = df['X']\n    return [\n        X.std(),\n    ]\n\ndef describe_y(df):\n    Y = df['Y']\n    return [\n        Y.std(),\n    ]\n\ndef describe_z(df):\n    Z = df['Z']\n    return [\n        Z.std(),  \n    ]\n\ndef describe_enmo(df):\n    enmo = df['enmo']\n    return [\n        enmo.mean(),  \n    ]\n\ndef describe_anglez(df):\n    anglez = df['anglez']\n    return [\n        anglez.std(),\n    ]\n    \n# Light level thresholds (in lux)\nlight_bins = [\n    (0, 5, 'Twilight'),\n    (5, 10, 'Minimal Street Lighting'),\n    (10, 50, 'Sunset'),\n    (50, 80, 'Family Living Room'),\n    (80, 100, 'Hallway'),\n    (100, 320, 'Very Dark Overcast Day'),\n    (320, 500, 'Office Lighting'),\n    (500, 1000, 'Sunrise/Sunset'),\n    (1000, 10000, 'Overcast Day'),\n    (10000, 25000, 'Full Daylight'),\n    (25000, 130000, 'Direct Sunlight')\n]\n\n\ndef categorize_light(light_value):\n    for low, high, label in light_bins:\n        if low <= light_value < high:\n            return label\n    return 'Unknown'\n\ndef describe_light(df):\n    df['light_category'] = df['light'].apply(categorize_light)\n    light_categories = df['light_category'].value_counts(normalize=True).to_dict()\n    \n    features = [light_categories.get(label, 0) for _, _, label in light_bins]\n    return features\n\ndef longest_inactivity_streaks(df, window_size=100, threshold=10, top_n=5):\n    rolling_cumsum = df['enmo'].rolling(window=window_size).sum()\n    inactive = rolling_cumsum <= threshold\n    \n    # Calculate streaks\n    streak_lengths = []\n    current_streak = 0\n    for is_inactive in inactive:\n        if is_inactive:\n            current_streak += 1\n        else:\n            if current_streak > 0:\n                streak_lengths.append(current_streak)\n            current_streak = 0\n    \n    # If the last streak is still active, add it\n    if current_streak > 0:\n        streak_lengths.append(current_streak)\n    \n    # Sort streaks in descending order and pick top N\n    streak_lengths = sorted(streak_lengths, reverse=True)[:top_n]\n    \n    # Pad with zeros if there are fewer than N streaks\n    streak_lengths += [0] * (top_n - len(streak_lengths))\n    return streak_lengths\n\n\ndef longest_activity_streaks(df, window_size=100, threshold=1, top_n=5):\n    # Calculate cumsum of enmo in the defined window\n    rolling_cumsum = df['enmo'].rolling(window=window_size).sum()\n    \n    # Identify active windows (cumsum > threshold)\n    active = rolling_cumsum > threshold\n    \n    # Calculate streaks\n    streak_lengths = []\n    current_streak = 0\n    for is_active in active:\n        if is_active:\n            current_streak += 1\n        else:\n            if current_streak > 0:\n                streak_lengths.append(current_streak)\n            current_streak = 0\n    \n    # If the last streak is still active, add it\n    if current_streak > 0:\n        streak_lengths.append(current_streak)\n    \n    # Sort streaks in descending order and pick top N\n    streak_lengths = sorted(streak_lengths, reverse=True)[:top_n]\n    \n    # Pad with zeros if there are fewer than N streaks\n    streak_lengths += [0] * (top_n - len(streak_lengths))\n    return streak_lengths\n\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop(['step'], axis=1, inplace=True)\n   \n    features = []\n    features.extend(describe_x(df))\n    features.extend(describe_y(df))\n    features.extend(describe_z(df))\n    features.extend(describe_enmo(df))\n    features.extend(describe_anglez(df))\n    features.extend(describe_light(df))  \n    \n    enmo_active_ratio = (df['enmo'] > 0).mean()\n    features.append(enmo_active_ratio)\n    features.extend(longest_inactivity_streaks(df, threshold=1))\n    features.extend(longest_activity_streaks(df, threshold=5))\n   \n    return np.array(features), filename.split('=')[1]\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:15:14.190175Z","iopub.execute_input":"2024-12-20T10:15:14.190635Z","iopub.status.idle":"2024-12-20T10:15:14.316188Z","shell.execute_reply.started":"2024-12-20T10:15:14.190588Z","shell.execute_reply":"2024-12-20T10:15:14.315014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-12-20T10:15:14.318910Z","iopub.execute_input":"2024-12-20T10:15:14.319308Z","iopub.status.idle":"2024-12-20T10:19:59.598930Z","shell.execute_reply.started":"2024-12-20T10:15:14.319275Z","shell.execute_reply":"2024-12-20T10:19:59.597782Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:19:59.600553Z","iopub.execute_input":"2024-12-20T10:19:59.601020Z","iopub.status.idle":"2024-12-20T10:19:59.643671Z","shell.execute_reply.started":"2024-12-20T10:19:59.600971Z","shell.execute_reply":"2024-12-20T10:19:59.642341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n\n    for col, (col_min, col_max) in min_max_dict.items():\n        df[col] = df[col].clip(lower=col_min, upper=col_max)\n\n    bins = [0, 6, 12, 18, 100]\n    labels = ['1 to 6', '7 to 12', '13 to 18', '19 to 100']\n    df['Age_Binned'] = pd.cut(df['Basic_Demos-Age'], bins=bins, labels=labels, right=True)\n    df['Age_Sex'] = df['Age_Binned'].astype(str) + '_' + df['Basic_Demos-Sex'].astype(str)\n    \n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    \n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    \n    df['PreInt_FGC_CU_PU'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_GSND_GSD_Age'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['SDS_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['SDS-SDS_Total_T']\n    \n    df['CGasync_Score_Normalized'] = df['CGAS-CGAS_Score'] - df.groupby('Basic_Demos-Enroll_Season')['CGAS-CGAS_Score'].transform('mean')\n    df['Internet_Physical_Difference'] = df['PreInt_EduHx-computerinternet_hoursday'] - df['PAQ_A-PAQ_A_Total']\n   \n    df[df.select_dtypes(include='object').columns] = df.select_dtypes(include='object').astype('category')\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:19:59.645078Z","iopub.execute_input":"2024-12-20T10:19:59.645545Z","iopub.status.idle":"2024-12-20T10:19:59.656696Z","shell.execute_reply.started":"2024-12-20T10:19:59.645484Z","shell.execute_reply":"2024-12-20T10:19:59.655442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n\nnumeric_cols = train[test.columns].select_dtypes(include='number').columns\nmin_max_dict = {col: (train[col].min(), train[col].max()) for col in numeric_cols}\n\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\ntrain = train.dropna(subset='sii')\n\ntarget = train['PCIAT-PCIAT_Total']\nsii_target = train['sii']\ntrain = train[test.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:19:59.658382Z","iopub.execute_input":"2024-12-20T10:19:59.658863Z","iopub.status.idle":"2024-12-20T10:20:00.022032Z","shell.execute_reply.started":"2024-12-20T10:19:59.658820Z","shell.execute_reply":"2024-12-20T10:20:00.020292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def map_pciat_to_sii(pciat_values):\n    return np.select(\n        [pciat_values <= 30, \n         (pciat_values > 30) & (pciat_values <= 49),\n         (pciat_values > 49) & (pciat_values <= 79),\n         pciat_values > 79],\n        [0, 1, 2, 3],\n        default=3  # For PCIAT values greater than 79\n    )\n    \ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:20:00.023475Z","iopub.execute_input":"2024-12-20T10:20:00.023825Z","iopub.status.idle":"2024-12-20T10:20:00.032275Z","shell.execute_reply.started":"2024-12-20T10:20:00.023793Z","shell.execute_reply":"2024-12-20T10:20:00.030821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_subset(df, target, subset_size=0.8):\n    df_subset = df.sample(frac=subset_size, random_state=42)\n    target_subset = target.loc[df_subset.index]\n    return df_subset, target_subset\n\n\ndef gaussian_noise_injection(df, target, noise_level, subset_size=0.2):\n\n    # Select a subset of data for augmentation\n    df_subset, target_subset = select_subset(df, target, subset_size)\n\n    # Split numeric and non-numeric columns\n    numeric_cols = df_subset.select_dtypes(include=['float64', 'int64'])\n    non_numeric_cols = df_subset.select_dtypes(exclude=['float64', 'int64'])\n\n    # Impute missing values in numeric columns\n    imputer = SimpleImputer(strategy='mean')\n    numeric_imputed = pd.DataFrame(imputer.fit_transform(numeric_cols), \n                                   columns=numeric_cols.columns, \n                                   index=numeric_cols.index)\n\n    # Add noise to numeric columns\n    augmented_numeric = numeric_imputed\n    for col in augmented_numeric.columns:\n        std_dev = augmented_numeric[col].std()\n        if std_dev > 0:  # Add noise only if variability exists\n            noise = np.random.normal(0, noise_level * std_dev, size=len(augmented_numeric))\n            augmented_numeric[col] += noise\n\n    # Concatenate back with non-numeric columns (align rows)\n    augmented_df = pd.concat([augmented_numeric, non_numeric_cols], axis=1)\n\n    # Ensure the column order matches the original subset\n    augmented_df = augmented_df[df_subset.columns]\n    return augmented_df, target_subset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:20:00.033665Z","iopub.execute_input":"2024-12-20T10:20:00.034182Z","iopub.status.idle":"2024-12-20T10:20:00.055465Z","shell.execute_reply.started":"2024-12-20T10:20:00.034130Z","shell.execute_reply":"2024-12-20T10:20:00.054200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def augment_data_with_nans(X, target, threshold=0.1, subset_size=0.2):\n   \n    df_subset, target_subset = select_subset(X, target, subset_size)\n    X_augmented = df_subset.reset_index(drop=True).copy()\n    \n    # Identify columns that already contain NaN values\n    columns_with_nan = [col for col in X.columns if X[col].isna().sum() > 0]\n    \n    # Mask for non-NaN values in columns that contain NaNs\n    non_nan_mask = X_augmented[columns_with_nan].notna()\n    \n    # Randomly select which column to set to NaN (for each row) where there's a valid value\n    for col in columns_with_nan:\n        # Create a random mask for columns with valid values (non-NaN)\n        random_mask = np.random.rand(len(X_augmented)) < threshold  # Adjust probability as needed\n        \n        # Apply the mask to select rows and set that column's value to NaN\n        X_augmented.loc[random_mask, col] = np.nan\n    \n    return X_augmented, target_subset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:20:00.058899Z","iopub.execute_input":"2024-12-20T10:20:00.059335Z","iopub.status.idle":"2024-12-20T10:20:00.074999Z","shell.execute_reply.started":"2024-12-20T10:20:00.059297Z","shell.execute_reply":"2024-12-20T10:20:00.073489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_confusion_matrix(y_true, y_pred, labels=None):\n    y_true = y_true.astype(np.int32)\n    y_pred = y_pred.astype(np.int32)\n    \n    if labels is None:\n        labels = sorted(set(y_true))\n\n    cm = confusion_matrix(y_true, y_pred, labels=labels)\n\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=labels)\n    disp.plot(cmap='Blues', values_format='d')\n\n    plt.title(\"Confusion Matrix\")\n    plt.xlabel(\"Predicted Label\")\n    plt.ylabel(\"True Label\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:20:00.076638Z","iopub.execute_input":"2024-12-20T10:20:00.077383Z","iopub.status.idle":"2024-12-20T10:20:00.099687Z","shell.execute_reply.started":"2024-12-20T10:20:00.077338Z","shell.execute_reply":"2024-12-20T10:20:00.098511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_validate_model(params, X, y, sii_target, label='', save_models=True, pruning_callback=None, n_repeats=5, return_qwk=False):\n    features = X.columns\n    start_time = time.time()\n    oof = []\n    y_oof = []\n    qwk_list = []\n    model_list = []\n   \n    n = 0\n    for repeat in tqdm(range(n_repeats)):\n        random_seed = np.random.randint(0, 10000)  # Generate a random seed for each repeat\n        folds = StratifiedKFold(n_splits=5, shuffle=True, random_state=repeat)\n        \n        for fold, (idx_tr, idx_va) in enumerate(folds.split(X, sii_target)):\n            params['random_seeds'] = n\n            set_global_seed(n)\n            X_tr = X.iloc[idx_tr]\n            X_va = X.iloc[idx_va]\n            y_tr = y.iloc[idx_tr]\n            y_va = y.iloc[idx_va]\n            \n            \n            nan_prone_columns = [\n                col for col in X_tr.columns \n                if X_tr[col].isna().any()  # Has NaNs\n            ]\n    \n            # Step 1: Perform augmentation on X_tr\n            nan_augmented, nan_aug_target = augment_data_with_nans(X_tr, target, threshold=1, subset_size=0.2)\n            noise_augmented, noise_aug_target = gaussian_noise_injection(X_tr, y_tr, noise_level=0.02, subset_size=0.5)\n    \n            X_tr_augmented = pd.concat(\n                [nan_augmented, noise_augmented, X_tr[y_tr>49], X_tr[y_tr>49], X_tr[y_tr>49], X_tr[y_tr>79]],\n                ignore_index=True).reset_index(drop=True)\n            \n            y_tr_augmented = pd.concat(\n                [nan_aug_target, noise_aug_target, y_tr[y_tr>49], y_tr[y_tr>49], y_tr[y_tr>49], y_tr[y_tr>79]],\n                ignore_index=True).reset_index(drop=True)\n\n\n            X_tr_combined = pd.concat([X_tr, X_tr_augmented], ignore_index=True).reset_index(drop=True)\n            y_tr_combined = pd.concat([y_tr, y_tr_augmented], ignore_index=True).reset_index(drop=True)\n\n            shuffled_indices = np.random.permutation(X_tr_combined.index)\n            X_tr_combined = X_tr_combined.iloc[shuffled_indices].reset_index(drop=True)\n            y_tr_combined = y_tr_combined.iloc[shuffled_indices].reset_index(drop=True)\n\n            \n            dtrain = lgb.Dataset(X_tr_combined, label=y_tr_combined)\n            dvalid = lgb.Dataset(X_va, label=y_va)\n\n            model = lgb.train(\n                params,\n                dtrain,\n                valid_sets=[dtrain, dvalid],\n                num_boost_round=params['n_estimators'],\n            )\n\n            y_pred = model.predict(X_va)\n\n            if save_models:\n                model_list.append(model)\n            oof.append(y_pred)\n            y_oof.append(y_va)\n            \n            n +=1\n    elapsed_time = time.time() - start_time\n\n    y_oof_actuals = np.concatenate(y_oof)\n    oof_preds = np.concatenate(oof)\n    \n    # Post-processing: Map predictions\n    y_oof_sii = map_pciat_to_sii(y_oof_actuals)\n    oof_sii = map_pciat_to_sii(oof_preds)\n\n  \n    qwk = cohen_kappa_score(y_oof_sii, oof_sii, weights='quadratic')\n    mse = ((y_oof_actuals - oof_preds)**2).mean()  \n    print(f\"Overall QWK: {qwk:.3f}, MSE: {mse:.3f}, Time: {int((time.time() - start_time) / 60)} min\")\n\n    # Optimize thresholds\n    threshold_optimizer = minimize(evaluate_predictions, \n                                   x0=[34, 49, 62], \n                                   args=(y_oof_sii, oof_preds), \n                                   method='Nelder-Mead')\n    \n    optimized_preds = threshold_Rounder(oof_preds, threshold_optimizer.x)\n    optimized_qwk = cohen_kappa_score(y_oof_sii, optimized_preds, weights='quadratic')\n    accuracy = (y_oof_sii==optimized_preds).astype(np.float32).mean()\n    print(f\"Optimized QWK: {optimized_qwk:.3f}, Accuracy: {accuracy:.3f}, Thresholds: {threshold_optimizer.x}\")\n    \n    plot_confusion_matrix(y_oof_sii, oof_sii)\n    \n    if save_models:\n        saved_models[label] = {'features': features, 'model_list': model_list}\n\n    return optimized_qwk, threshold_optimizer.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:20:00.115648Z","iopub.execute_input":"2024-12-20T10:20:00.116022Z","iopub.status.idle":"2024-12-20T10:20:00.134339Z","shell.execute_reply.started":"2024-12-20T10:20:00.115982Z","shell.execute_reply":"2024-12-20T10:20:00.133049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"saved_models = {}\nresults = []\nfor i in range(1):\n    params = {'verbosity': -1,  'device': 'cpu', 'metric': 'mse', 'n_estimators':150, 'max_depth':5, 'max_bin': 15, 'boosting_type': 'gbdt', 'lambda_l1': 0.0012071403780584485, 'lambda_l2': 19.943477818207878, 'min_child_weight': 0.01586977190723854, 'learning_rate': 0.030512450456770007, 'num_leaves': 295, 'colsample_bytree': 0.8569995659929517, 'bagging_fraction': 0.587037100215173, 'feature_fraction': 0.8955475330753205, 'bagging_freq': 1}\n    qwk, qwk_thresholded = cross_validate_model(params, train, target, sii_target, label='trial', save_models=True, n_repeats=100)\n    print(qwk)\n    results.append(qwk)\nprint(f\"'mean {np.mean(results)}\")\nprint(f\"diff {max(results) - min(results)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:36:25.386001Z","iopub.execute_input":"2024-12-20T10:36:25.386426Z","iopub.status.idle":"2024-12-20T10:47:04.429433Z","shell.execute_reply.started":"2024-12-20T10:36:25.386387Z","shell.execute_reply":"2024-12-20T10:47:04.428045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = [model.predict(test)  for model in saved_models['trial']['model_list']]\n\nn = 16\ni = 500\nplt.hist(np.array(pred)[:, n][:i], bins=30, alpha=0.7)\n\n# Get the mode\nmode_val = stats.mode(np.array(pred)[:, n][:i].round())[0]  # mode.value[0]\n\n# Overlay the mode on the histogram\nplt.axvline(mode_val, color='k', linestyle='dashed', linewidth=2, label=f'Mode: {mode_val}')\nplt.axvline(np.array(pred)[:, n][:i].mean(), color='r', linestyle='dashed', linewidth=2, label=f'mean: {np.array(pred)[:, n][:i].mean()}')\n# Add a label\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:47:34.345600Z","iopub.execute_input":"2024-12-20T10:47:34.346024Z","iopub.status.idle":"2024-12-20T10:47:40.011193Z","shell.execute_reply.started":"2024-12-20T10:47:34.345988Z","shell.execute_reply":"2024-12-20T10:47:40.009829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = stats.mode(threshold_Rounder(np.array([model.predict(test) for model in saved_models['trial']['model_list']]), qwk_thresholded).astype(np.int32))[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:47:43.829032Z","iopub.execute_input":"2024-12-20T10:47:43.829778Z","iopub.status.idle":"2024-12-20T10:47:49.145168Z","shell.execute_reply.started":"2024-12-20T10:47:43.829734Z","shell.execute_reply":"2024-12-20T10:47:49.144015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nsubmission_df['sii'] = predictions\nsubmission_df.to_csv('submission.csv', index=False)\npd.read_csv('./submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:47:49.147035Z","iopub.execute_input":"2024-12-20T10:47:49.147418Z","iopub.status.idle":"2024-12-20T10:47:49.170821Z","shell.execute_reply.started":"2024-12-20T10:47:49.147382Z","shell.execute_reply":"2024-12-20T10:47:49.169187Z"}},"outputs":[],"execution_count":null}]}