{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"text-align:center\"><span style=\"background-color:#a8edc4;color:black;padding:10px;border-radius:40px;\">🌐 AutoML H2O</span></div>\n\n\n![](https://i.postimg.cc/Jz5DS0dq/pexels-james-frid-81279-9823161.jpg)\n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">❗Understanding the Competition</span>\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h3 style=\"color:black;\">Aim of competition</h3>\n  <p style=\"color:black;\">This competition aimed to predict the severity of Problematic Internet Use (PIU) in children and adolescents based on their physical activity and fitness data.</p>\n\n  <h2 style=\"color:black;\">Key Points:</h2>\n  <ul>\n    <li><strong>Data</strong>: The competition utilized data from the Healthy Brain Network study, including:\n      <ul>\n        <li>Wrist-worn accelerometer data</li>\n        <li>Fitness assessments</li>\n        <li>Questionnaires</li>\n      </ul>\n    </li>\n    <li><strong>Target Variable</strong>: Severity Impairment Index (SII), a standard measure of PIU.</li>\n    <li><strong>Challenge</strong>: The dataset had missing values, particularly for the target variable SII</li>\n    <li><strong>Impact</strong>: Successful models could potentially help identify individuals at risk of PIU and enable early intervention strategies.</li>\n  </ul>\n</div>\n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🎒 Import Libraries</span>","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nimport os\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nimport h2o\nfrom h2o.automl import H2OAutoML\nfrom sklearn.base import clone\nfrom sklearn.metrics import *\nfrom colorama import Fore, Style\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T18:01:41.584197Z","iopub.execute_input":"2024-10-30T18:01:41.584664Z","iopub.status.idle":"2024-10-30T18:01:41.592792Z","shell.execute_reply.started":"2024-10-30T18:01:41.584622Z","shell.execute_reply":"2024-10-30T18:01:41.591528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Python Modules</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Imported Modules</h2>\n  <ul>\n    <li><strong>NumPy (np):</strong> For numerical computations</li>\n    <li><strong>Pandas (pd):</strong> For data analysis and manipulation</li>\n    <li><strong>Seaborn (sns):</strong> For visualization</li>\n    <li><strong>matplotlib.pyplot (plt):</strong> For plotting</li>\n    <li><strong>StratifiedKFold (sklearn.model_selection):</strong> For stratified K-Fold cross-validation</li>\n    <li><strong>SciPy.optimize (scipy.optimize):</strong> For optimization routines</li>\n    <li><strong>OS:</strong> For operating system interactions</li>\n    <li><strong>TQDM:</strong> For progress bar visualization</li>\n    <li><strong>IPython.display:</strong> For displaying output in Jupyter notebooks</li>\n    <li><strong>Concurrent.futures (concurrent.futures):</strong> For parallel execution of tasks</li>\n    <li><strong>StandardScaler (sklearn.preprocessing):</strong> For feature standardization</li>\n    <li><strong>LabelEncoder (sklearn.preprocessing):</strong> For encoding categorical features</li>\n    <li><strong>H2O (h2o):</strong> For distributed machine learning platform</li>\n    <li><strong>Sklearn.base (sklearn.base):</strong> For base classes for estimators</li>\n    <li><strong>Sklearn.metrics (sklearn.metrics):</strong> For calculating metrics</li>\n    <li><strong>Colorama (colorama):</strong> For colored text output</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Global Variables</h2>\n  <ul>\n    <li><strong>SEED (integer):</strong> Random seed for reproducibility</li>\n    <li><strong>n_splits (integer):</strong> Number of folds</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📤 Load CSV Files</span>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nTARGET_COLS = [\n    \"PCIAT-Season\",\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",    \n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n    \"PCIAT-PCIAT_Total\"\n]\n\ntrain_df = train_df.drop(TARGET_COLS,axis=1)\n\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nids = test_df['id']\n\nSEASON_COLS = [\n    \"Basic_Demos-Enroll_Season\", \"CGAS-Season\", \"Physical-Season\",\n    \"Fitness_Endurance-Season\", \"FGC-Season\", \"BIA-Season\",\n    \"PAQ_A-Season\", \"PAQ_C-Season\", \"SDS-Season\", \"PreInt_EduHx-Season\"\n]\n\ntrain_df = train_df.drop(SEASON_COLS,axis=1)\ntest_df = test_df.drop(SEASON_COLS,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:01:41.638208Z","iopub.execute_input":"2024-10-30T18:01:41.638750Z","iopub.status.idle":"2024-10-30T18:01:41.709540Z","shell.execute_reply.started":"2024-10-30T18:01:41.638698Z","shell.execute_reply":"2024-10-30T18:01:41.707882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Load CSV Files</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black !important;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Data Loading</h2>\n  <ul>\n    <li><strong>train_df</strong> is loaded from the specified CSV file containing the training data.</li>\n    <li><strong>sample</strong> is loaded from the specified CSV file containing the sample submission data.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Target Columns</h2>\n  <ul><li>A list of target columns (<strong>TARGET_COLS</strong>) is defined, containing the PCIAT columns that will be removed.</li></ul>\n\n<h2 style=\"color:black;\">Dropping Target Columns</h2>\n<ul>\n    <li>The target columns are dropped from the training data (<strong>train_df</strong>) to remove features that are not in test data.</li></ul>\n\n<h2 style=\"color:black;\">Loading Test Data and Extracting IDs</h2>\n<ul><li>The test data (<strong>test_df</strong>) is loaded from the test CSV file.</li> \n    <li>The \"id\" column is extracted from the test data and stored in the <strong>ids</strong> variable.</li></ul>\n<h2 style=\"color:black;\">Season Columns</h2>\n  <ul><li>A list of season columns (<strong>SEASON_COLS</strong>) is defined, containing the season columns that will be removed.</li></ul>\n\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📤 Load Parquet Files</span>","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:01:41.711898Z","iopub.execute_input":"2024-10-30T18:01:41.712443Z","iopub.status.idle":"2024-10-30T18:03:35.477564Z","shell.execute_reply.started":"2024-10-30T18:01:41.712387Z","shell.execute_reply":"2024-10-30T18:03:35.476196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Load Parquet Files</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n\n  <h2 style=\"color:black;\">Process Function (<strong>process_file</strong>)</h2>\n\n  <ul>\n    <li>Reads the Parquet file using <strong>pd.read_parquet</strong>.</li>\n    <li>Drops the \"step\" column (assuming it's not relevant for the analysis).</li>\n    <li>Calculates descriptive statistics for the remaining columns and reshapes the results into a one-dimensional array.</li>\n    <li>Extracts the ID from the filename (assuming the format filename=ID.parquet) and returns a tuple containing the statistics and the ID.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Load Time Series (<strong>load_time_series</strong>)</h2>\n\n  <ul>\n    <li>It lists all files (assumed to be time series data) in the directory using <strong>os.listdir(dirname)</strong>. </li>\n    <li>It utilizes a <strong>ThreadPoolExecutor</strong> for parallel processing to improve efficiency.</li>\n    <ul>\n      <li>The <strong>executor.map</strong> function applies the <strong>process_file</strong> function to each filename asynchronously.</li>\n      <li>A progress bar (<strong>tqdm</strong>) is used to visualize the progress of processing files.</li>\n    </ul>\n    <li>It unpacks the results obtained from parallel processing into separate lists for statistics (<strong>stats</strong>) and IDs (<strong>indexes</strong>).</li>\n    <li>It creates a DataFrame (<strong>df</strong>) from the statistics, with column names generated as \"Stat_i\" for each statistic index (<strong>i</strong>).</li>\n    <li>It adds a new column named \"id\" to the DataFrame containing the extracted IDs.</li>\n    <li>Finally, it returns the DataFrame containing the processed time series data.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Loading Training and Test Time Series</h2>\n  <ul>\n    <li>The <strong>load_time_series</strong> function is used to load time series data from training and test directories, respectively.</li> \n        <li>The resulting DataFrames are stored in <strong>train_ts</strong> and <strong>test_ts</strong>.</li>\n    <li>Time series feature columns are extracted from the training data columns (<strong>time_series_cols</strong>) by removing the \"id\" column.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🗒️ Merge CSV & Parquet</span>","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train_df, train_ts, how=\"left\", on='id')\ntest = pd.merge(test_df, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\ntrain = train.dropna(subset=['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.479651Z","iopub.execute_input":"2024-10-30T18:03:35.480186Z","iopub.status.idle":"2024-10-30T18:03:35.509557Z","shell.execute_reply.started":"2024-10-30T18:03:35.480138Z","shell.execute_reply":"2024-10-30T18:03:35.508238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Merge CSV & Parquet</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Merging Tabular and Time Series Data</h2>\n  <ul>\n    <li><strong>train_df</strong> and <strong>train_ts</strong> are merged based on the \"id\" column using a left join.</li>\n    <li><strong>test_df</strong> and <strong>test_ts</strong> are merged based on the \"id\" column using a left join.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Dropping Unnecessary Column</h2>\n  <ul>\n    <li>The \"id\" column is dropped from both the training and testing DataFrames as it is no longer needed for model training.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Handling Missing Values</h2>\n  <ul>\n    <li>Rows with missing values in the \"sii\" column (i.e target variable) are dropped from the training DataFrame.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">✨ Preprocessing</span>","metadata":{}},{"cell_type":"code","source":"def preprocess_data(df,train_data=False):\n    # Handle numerical columns\n    num_cols = df.select_dtypes(include=np.number).columns\n    df[num_cols] = df[num_cols].fillna(df[num_cols].median())\n     \n    return df\n\ntrain = preprocess_data(train)\ntest = preprocess_data(test)\n\ntrain = pd.DataFrame(train)\ntest = pd.DataFrame(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.511308Z","iopub.execute_input":"2024-10-30T18:03:35.511708Z","iopub.status.idle":"2024-10-30T18:03:35.696445Z","shell.execute_reply.started":"2024-10-30T18:03:35.511667Z","shell.execute_reply":"2024-10-30T18:03:35.695244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Preprocessing</span>\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Data Preprocessing and Feature Engineering</h2>\n\n  <p style=\"color:black;\">This code snippet defines a function: <strong>preprocess_data</strong> to handle features, respectively.</p>\n\n  <h3 style=\"color:black;\">Preprocessing Data</h3>\n  <p style=\"color:black;\">The <strong>preprocess_data</strong> function performs the following steps:</p>\n  <ul>\n    <li><strong>Handling Numerical Features</strong>\n      <ul>\n        <li>Identifies numerical columns using <strong>df.select_dtypes(include=np.number).columns</strong>.</li>\n        <li>Fills missing values in numerical columns with the median value using <strong>df[num_cols].fillna(df[num_cols].median())</strong>.</li>\n      </ul>\n    </li>\n  </ul>\n\n  <p style=\"color:black;\">Finally, the preprocessed training and testing DataFrames are converted back to Pandas DataFrames.</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🔎 View Data</span>","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.709137Z","iopub.execute_input":"2024-10-30T18:03:35.709661Z","iopub.status.idle":"2024-10-30T18:03:35.760818Z","shell.execute_reply.started":"2024-10-30T18:03:35.709605Z","shell.execute_reply":"2024-10-30T18:03:35.759445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.762610Z","iopub.execute_input":"2024-10-30T18:03:35.763042Z","iopub.status.idle":"2024-10-30T18:03:35.802397Z","shell.execute_reply.started":"2024-10-30T18:03:35.762986Z","shell.execute_reply":"2024-10-30T18:03:35.800888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: View Data</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Displaying First Few Rows of DataFrames</h2>\n  <ul>\n    <li>Use the <strong>head()</strong> method to display the first few rows of the preprocessed training and testing DataFrames.</li>\n    <li>This provides insights into the data structure, column names, and values.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">⚖️ Quadratic Weighted Kappa</span>","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data, train_data):\n    X = train_data.drop(['sii'], axis=1)\n    y = train_data['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        X_train = h2o.H2OFrame(X_train)\n        y_train_pred = best_model.predict(X_train)\n        y_train_pred = y_train_pred.as_data_frame()\n        y_train_pred = y_train_pred.values\n        y_train_pred = y_train_pred.reshape(-1)\n        \n        X_val = h2o.H2OFrame(X_val)\n        y_val_pred = best_model.predict(X_val)\n        y_val_pred = y_val_pred.as_data_frame()\n        y_val_pred = y_val_pred.values\n        y_val_pred = y_val_pred.reshape(-1)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_data_copy = h2o.H2OFrame(test_data)\n        test_data_copy = best_model.predict(test_data_copy)\n        test_data_copy = test_data_copy.as_data_frame()\n        test_data_copy = test_data_copy.values\n        test_data_copy = test_data_copy.reshape(-1)\n        test_preds[:, fold] = test_data_copy        \n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,KappaOPtimizer","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.804823Z","iopub.execute_input":"2024-10-30T18:03:35.805307Z","iopub.status.idle":"2024-10-30T18:03:35.829373Z","shell.execute_reply.started":"2024-10-30T18:03:35.805255Z","shell.execute_reply":"2024-10-30T18:03:35.826169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Quadratic Weighted Kappa</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Quadratic Weighted Kappa and Training Function</h2>\n\n  <p style=\"color:black;\">This section defines two functions: <strong>quadratic_weighted_kappa</strong> and <strong>TrainML</strong>.</p>\n\n  <h3 style=\"color:black;\">quadratic_weighted_kappa</h3>\n  <ul><li>This function calculates the quadratic weighted kappa score between two sets of labels (<strong>y_true</strong> and <strong>y_pred</strong>).</li><li> It utilizes the <strong>cohen_kappa_score</strong> function from scikit-learn with the <strong>weights='quadratic'</strong> argument.</li></ul>\n\n  <h3 style=\"color:black;\">TrainML</h3>\n  <p style=\"color:black;\">This function is responsible for training a machine learning model and evaluating its performance. It takes the model class (<strong>model_class</strong>), testing data (<strong>test_data</strong>), and training data (<strong>train_data</strong>) as input.</p>\n\n  <ul>\n    <li>It separates the target variable (<strong>sii</strong>) from the features in the training data (<strong>X</strong> and <strong>y</strong>).</li>\n    <li>It uses <strong>StratifiedKFold</strong> to perform stratified K-Fold cross-validation, ensuring balanced class distribution in each fold.</li>\n    <li>It initializes arrays to store training and validation kappa scores (<strong>train_S</strong>, <strong>test_S</strong>), predicted probabilities (<strong>oof_non_rounded</strong>), rounded predictions (<strong>oof_rounded</strong>), and test set predictions (<strong>test_preds</strong>).</li>\n    <li>It iterates through each fold using a progress bar (<strong>tqdm</strong>).</li>\n      <ul>\n        <li>Within each fold, it splits the training data into training and validation sets (<strong>X_train</strong>, <strong>X_val</strong>, <strong>y_train</strong>, <strong>y_val</strong>).</li>\n        <li>It converts the training and validation sets to H2O Frames.</li>\n        <li>It uses the <strong>best_model</strong> to predict probabilities for both the training and validation sets.</li>\n        <li>It converts the H2OFrame predictions to NumPy arrays.</li>\n        <li>It stores the predicted probabilities for the validation set (<strong>oof_non_rounded</strong>) and rounds them to integers for evaluation (<strong>oof_rounded</strong>).</li>\n        <li>It calculates and stores the quadratic weighted kappa scores for the training and validation sets (<strong>train_kappa</strong>, <strong>val_kappa</strong>).</li>\n        <li>It appends these kappa scores to their respective lists (<strong>train_S</strong>, <strong>test_S</strong>).</li>\n        <li>It converts the test data to an H2O Frame.</li>\n        <li>It uses the <strong>best_model</strong> to predict probabilities for the test data.</li>\n        <li>It converts the H2OFrame predictions to a NumPy array and stores them in <strong>test_preds[:, fold]</strong>.</li>\n        <li>It prints the fold number, training kappa score, and validation kappa score.</li>\n      </ul>\n    <li>It prints the mean training and validation kappa scores.</li>\n    <li>It performs optimization using <strong>minimize</strong> to find the optimal thresholds for rounding the predicted probabilities. This optimization minimizes the negative quadratic weighted kappa (<strong>evaluate_predictions</strong>) on the validation set (<strong>y</strong>, <strong>oof_non_rounded</strong>). The Nelder-Mead method is used for optimization.</li>\n    <li>It asserts that the optimization converged successfully.</li>\n    <li>It applies the optimized thresholds to the validation set probabilities (<strong>oof_tuned</strong>) and calculates the final kappa score (<strong>tKappa</strong>).</li>\n    <li>It prints the optimized kappa score.</li>\n    <li>It calculates the mean of the predicted probabilities across all folds for the test set (<strong>tpm</strong>).</li>\n    <li>It applies the optimized thresholds to the mean test set probabilities (<strong>tpTuned</strong>).</li>\n    <li>Finally, it creates a submission DataFrame with the IDs from the sample data and the rounded predictions (<strong>tpTuned</strong>) and returns this DataFrame along with the trained model.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🌐 AutoML H2O</span>","metadata":{}},{"cell_type":"code","source":"h2o.init()\ntrain_data = h2o.H2OFrame(train)\n\naml = H2OAutoML(max_runtime_secs=5400,seed=5)\naml.train(y='sii', training_frame=train_data)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:03:35.832495Z","iopub.execute_input":"2024-10-30T18:03:35.833186Z","iopub.status.idle":"2024-10-30T18:04:08.577781Z","shell.execute_reply.started":"2024-10-30T18:03:35.833122Z","shell.execute_reply":"2024-10-30T18:04:08.576438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: AutoML H2O</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Initializing H2O and Training an AutoML Model</h2>\n\n  <p style=\"color:black;\">This code snippet initializes the H2O environment and trains an AutoML model on the training data.</p>\n\n  <ul>\n    <li><strong>h2o.init()</strong>: Initializes the H2O cluster.</li>\n    <li><strong>train_data = h2o.H2OFrame(train)</strong>: Converts the Pandas DataFrame <strong>train</strong> into an H2OFrame for use with H2O algorithms.</li>\n    <li><strong>aml = H2OAutoML(max_runtime_secs, seed)</strong>: Creates an AutoML instance with a maximum runtime in seconds and a random seed of 5.</li>\n    <li><strong>aml.train(y='sii', training_frame=train_data)</strong>: Trains the AutoML model on the <strong>train_data</strong> with <strong>sii</strong> as the target variable.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🏆 Leaderboard</span>","metadata":{}},{"cell_type":"code","source":"leaderboard = aml.leaderboard","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:04:08.579420Z","iopub.execute_input":"2024-10-30T18:04:08.579799Z","iopub.status.idle":"2024-10-30T18:04:08.586287Z","shell.execute_reply.started":"2024-10-30T18:04:08.579759Z","shell.execute_reply":"2024-10-30T18:04:08.584751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Leaderboard</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Retrieving the AutoML Leaderboard</h2>\n\n  <ul><li>The <strong>leaderboard</strong> attribute of the AutoML object provides a DataFrame containing the performance metrics of the different models trained during the AutoML process.</li></ul>\n\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🥇 Best Models</span>","metadata":{}},{"cell_type":"code","source":"best_model = aml.leader\nSubmission,KappaOPtimizer = TrainML(best_model,test,train)\nprint(KappaOPtimizer.x)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:04:08.590550Z","iopub.execute_input":"2024-10-30T18:04:08.591117Z","iopub.status.idle":"2024-10-30T18:04:18.731388Z","shell.execute_reply.started":"2024-10-30T18:04:08.591056Z","shell.execute_reply":"2024-10-30T18:04:18.730067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Best Models</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Selecting the Best Model and Making Predictions</h2>\n\n  <ul>\n    <li><strong>best_model = aml.leader</strong>: Selects the best performing model from the AutoML experiment.</li>\n    <li><strong>Submission, KappaOPtimizer = TrainML(best_model, test, train)</strong>: Calls the <strong>TrainML</strong> function with the best model, test data, and training data to make predictions on the test set and optimize the threshold for the final predictions.</li>\n    <li><strong>print(KappaOPtimizer.x)</strong>: Prints the optimized threshold values.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📁 Prepare Submission CSV</span>","metadata":{}},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:04:18.732898Z","iopub.execute_input":"2024-10-30T18:04:18.733324Z","iopub.status.idle":"2024-10-30T18:04:18.741417Z","shell.execute_reply.started":"2024-10-30T18:04:18.733273Z","shell.execute_reply":"2024-10-30T18:04:18.740054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: Prepare Submission CSV</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Creating the Submission File</h2>\n\n  <ul><li>This step involves creating a CSV file with the predicted values for the test data.</li></ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📰 View Results</span>","metadata":{}},{"cell_type":"code","source":"print(Submission['sii'].value_counts())\nSubmission.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:04:18.743371Z","iopub.execute_input":"2024-10-30T18:04:18.743768Z","iopub.status.idle":"2024-10-30T18:04:18.769794Z","shell.execute_reply.started":"2024-10-30T18:04:18.743725Z","shell.execute_reply":"2024-10-30T18:04:18.768440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑 Explaining: View Results</span>\n\n<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Analyzing and Visualizing the Submission</h2>\n  <ul>\n    <li>Print the frequency of each predicted value in the \"sii\" column using <strong>Submission['sii'].value_counts()</strong>.</li>\n    <li>Display the first 20 rows of the submission DataFrame using <strong>Submission.head(20)</strong>.</li>\n  </ul>\n</div>","metadata":{}}]}