{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#https://www.kaggle.com/code/shahzaibmalik44/ensemble-learning-for-beginner-to-advance i copied it.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-13T02:57:25.744751Z","iopub.execute_input":"2024-11-13T02:57:25.745338Z","iopub.status.idle":"2024-11-13T02:57:25.770791Z","shell.execute_reply.started":"2024-11-13T02:57:25.745292Z","shell.execute_reply":"2024-11-13T02:57:25.769741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n!pip install lightgbm\n!pip install xgboost\n!pip install catboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T03:00:17.000627Z","iopub.execute_input":"2024-11-13T03:00:17.000953Z","iopub.status.idle":"2024-11-13T03:00:52.577359Z","shell.execute_reply.started":"2024-11-13T03:00:17.000918Z","shell.execute_reply":"2024-11-13T03:00:52.576207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Standard libraries  \nimport warnings  \nimport gc  \n\n# Data manipulation and analysis  \nimport numpy as np  \nimport pandas as pd  \n\n# Progress bar  \nfrom tqdm import tqdm  \n\n# Machine learning model selection and evaluation  \nfrom sklearn.model_selection import StratifiedKFold  \nfrom sklearn.base import clone  \n\n# Optimization  \nfrom scipy.optimize import minimize  \n\n# Machine learning models  \nfrom lightgbm import LGBMRegressor  \nfrom xgboost import XGBRegressor  \nfrom catboost import CatBoostRegressor  \n\n# Ensemble methods  \nfrom sklearn.ensemble import VotingRegressor  \n\n# Suppress warnings  \nwarnings.filterwarnings(\"ignore\")  \n\n# Garbage collection  \ngc.collect() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T03:01:16.692569Z","iopub.execute_input":"2024-11-13T03:01:16.692966Z","iopub.status.idle":"2024-11-13T03:01:20.927643Z","shell.execute_reply.started":"2024-11-13T03:01:16.692925Z","shell.execute_reply":"2024-11-13T03:01:20.926699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfrom concurrent.futures import ThreadPoolExecutor\nimport os\nfrom tqdm import tqdm\n\nenccoding_dim = 128\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Load datasets using context managers\ndef load_datasets():\n    try:\n        # Load CSV files\n        Train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n        Test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n        Data_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n        sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n        \n        print(\"Datasets loaded successfully.\")\n        return Train, Test, Data_dict, sample, train_ts, test_ts\n    \n    except Exception as e:\n        print(f\"Error loading datasets: {e}\")\n        gc.collect()\n        return None\n\n# Display dataset information\ndef display_dataset_info(dataset, name):\n    print(f\"{name} DataFrame Shape: Rows = {dataset.shape[0]}, Columns = {dataset.shape[1]}\")\n    \n    total_missing = dataset.isnull().sum().sum()\n    print(f\"There are {total_missing} missing values in the {name} DataFrame.\") if total_missing > 0 else print(f\"There are no missing values in the {name} DataFrame.\")\n\n    total_duplicates = dataset.duplicated().sum()\n    print(f\"There are {total_duplicates} duplicate rows in the {name} DataFrame.\") if total_duplicates > 0 else print(f\"There are no duplicate rows in the {name} DataFrame.\")\n    \n    display(dataset.describe().style.set_caption(f\"Descriptive Statistics for {name} Dataset\").background_gradient(cmap=\"coolwarm\"))\n    print(\"-----------------------------------------------------------------\")\n    gc.collect()\n\n# Function for memory optimization\ndef reduce_memory_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f\"Memory usage of dataframe is {start_mem:.2f} MB\")\n\n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        # Optimize numerical columns\n        if col_type != object and not pd.api.types.is_categorical_dtype(df[col]):\n            c_min, c_max = df[col].min(), df[col].max()\n            if str(col_type).startswith('int'):\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                else:\n                    df[col] = df[col].astype(np.int64)\n            elif str(col_type).startswith('float'):\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f\"{col} column memory usage after optimization is {end_mem:.2f} MB\")\n    print(f\"Decreased by {(100 * (start_mem - end_mem) / start_mem):.1f}%\\n\")\n    gc.collect()\n    return df\n\n# Load datasets\nTrain, Test, Data_dict, sample, train_ts, test_ts = load_datasets()\n\n# Display information for all datasets\ndatasets = [sample, Data_dict, Train, Test, train_ts, test_ts]\nnames = [\"Sample\", \"Data Dict\", \"Train\", \"Test\", \"Train Time Series\", \"Test Time Series\"]\nfor i in range(len(datasets)):\n    display_dataset_info(datasets[i], names[i])\n    \n\n# Reduce memory usage of the datasets\ntest_ts = reduce_memory_usage(test_ts)\ntrain_ts = reduce_memory_usage(train_ts)\n\nData_dict = reduce_memory_usage(Data_dict)\nsample = reduce_memory_usage(sample)\n\ntest = reduce_memory_usage(Test)\ntrain = reduce_memory_usage(Train)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T03:01:55.126127Z","iopub.execute_input":"2024-11-13T03:01:55.127250Z","iopub.status.idle":"2024-11-13T03:01:55.669457Z","shell.execute_reply.started":"2024-11-13T03:01:55.127207Z","shell.execute_reply":"2024-11-13T03:01:55.668598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.preprocessing import StandardScaler\n\n# Define the AutoEncoder class\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        \n        # Encoder layers\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim * 3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 3, encoding_dim * 2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 2, encoding_dim),\n            nn.ReLU()\n        )\n        \n        # Decoder layers\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim * 2),\n            nn.ReLU(),\n            nn.Linear(input_dim * 2, input_dim * 3),\n            nn.ReLU(),\n            nn.Linear(input_dim * 3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n# Function to perform autoencoding on a dataset\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    # Standardize the dataset\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    # Convert the scaled data to a tensor\n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    # Get input dimension\n    input_dim = data_tensor.shape[1]\n    \n    # Initialize the AutoEncoder model\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    # Define the loss function and optimizer\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    # Training loop\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            \n            # Forward pass\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            \n            # Backward pass and optimization\n            loss.backward()\n            optimizer.step()\n        \n        # Print loss every 10 epochs\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n\n    # Extract the encoded data\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n    \n    # Convert the encoded data to a DataFrame\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\n\ntrain_ts_encoded = perform_autoencoder(train_ts.drop('id', axis=1), encoding_dim=60, epochs=100, batch_size=32)\n\ntest_ts_encoded = perform_autoencoder(test_ts.drop('id', axis=1), encoding_dim=60, epochs=100, batch_size=32)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Merge the training and testing datasets with their respective time series data\n# This assumes that train_ts and test_ts dataframes are defined and contain a common 'id' column\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import LabelEncoder \nfrom sklearn.impute import KNNImputer  \n\n\ncategorical_cols = train.select_dtypes(include=[\"category\", \"object\"]).columns  \nnumeric_cols = train.select_dtypes(include=[\"number\"]).columns  \n\nlabel_encoder = LabelEncoder()  \nfor col in categorical_cols:\n    train[col] = label_encoder.fit_transform(train[col])     \n\n    \n# Create a KNN imputer instance  \nknn_imputer = KNNImputer(n_neighbors=7)  \n# Fit and transform the data  \nimputed_data = knn_imputer.fit_transform(train[numeric_cols])  \n# Create a DataFrame with the imputed data  \ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)  \n# Round and convert the 'sii' column to integer, if it exists  \nif 'sii' in train_imputed.columns:  \n    train_imputed['sii'] = train_imputed['sii'].round().astype(int)  \n    \ntrain[numeric_cols] = train_imputed[numeric_cols]\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import LabelEncoder \nfrom sklearn.impute import KNNImputer  \n\n\ncategorical_cols = train.select_dtypes(include=[\"category\", \"object\"]).columns  \nnumeric_cols = train.select_dtypes(include=[\"number\"]).columns  \n\nlabel_encoder = LabelEncoder()  \nfor col in categorical_cols:\n    train[col] = label_encoder.fit_transform(train[col])     \n\n    \n# Create a KNN imputer instance  \nknn_imputer = KNNImputer(n_neighbors=7)  \n# Fit and transform the data  \nimputed_data = knn_imputer.fit_transform(train[numeric_cols])  \n# Create a DataFrame with the imputed data  \ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)  \n# Round and convert the 'sii' column to integer, if it exists  \nif 'sii' in train_imputed.columns:  \n    train_imputed['sii'] = train_imputed['sii'].round().astype(int)  \n    \ntrain[numeric_cols] = train_imputed[numeric_cols]\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, filter out columns that match the regex pattern in test\ncolumns_to_drop = list(train.filter(regex='^stat_|PCIAT').columns) + [\n    'Physical-Season', 'PAQ_A-Season', 'BIA-Season', 'SDS-Season', \n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'PreInt_EduHx-Season', \n    'Fitness_Endurance-Season', 'PAQ_C-Season', 'FGC-Season', 'id']\n\nfeaturesCols = train.drop(columns = columns_to_drop).columns.tolist()\n\nfeaturesCols += train_ts.columns.tolist()\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# second, filter out columns that match the regex pattern in test\ncolumns_to_drop = list(test.filter(regex='^stat_|PCIAT').columns) + [\n    'Physical-Season', 'PAQ_A-Season', 'BIA-Season', 'SDS-Season', \n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'PreInt_EduHx-Season', \n    'Fitness_Endurance-Season', 'PAQ_C-Season', 'FGC-Season', 'id']\n\nfeaturesCols = test.drop(columns = columns_to_drop).columns.tolist()\n\nfeaturesCols += train_ts.columns.tolist()\ntest = test[featuresCols]\n\ntest = test.drop(columns = 'id')\ntrain = train.drop(columns = \"id\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create mapping for categorical columns\ndef create_mapping(column, dataset):\n    # Get unique values from the specified column in the dataset\n    unique_values = dataset[column].unique()\n    # Create a mapping where each unique value is assigned an index\n    return {value: idx for idx, value in enumerate(unique_values)}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score\n\n# Define quadratic weighted kappa metric\ndef quadratic_weighted_kappa(y_true, y_pred):\n    # Calculate Cohen's Kappa score with quadratic weights\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Threshold rounding function\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Function to train a machine learning model using cross-validation\ndef TrainML(model_class, test_data):\n    # Prepare features and target variable from the training set\n    X = train.drop(['sii'], axis=1)  # Features\n    y = train['sii']                  # Target variable\n\n    n_splits = 5  # Number of splits for cross-validation\n    SEED = 42     # Random seed for reproducibility\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    # Lists to store training and validation performance metrics\n    train_scores = []  \n    validation_scores = []  \n\n    # Arrays to store out-of-fold and test predictions\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    # Cross-validation loop\n    for fold, (train_idx, val_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        # Initialize and train the model\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        # Make predictions on training and validation sets\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # Store predictions for out-of-fold analysis\n        oof_non_rounded[val_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        # Calculate quadratic weighted kappa for both training and validation sets\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        # Append kappa scores to their respective lists\n        train_scores.append(train_kappa)\n        validation_scores.append(val_kappa)\n\n        # Store predictions for the test set\n        test_preds[:, fold] = model.predict(test_data)\n        \n        # Print fold results\n        print(\"-----------------------------------------\")\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    # Print mean kappa scores across all folds\n    print(f\"Mean Train QWK: {np.mean(train_scores):.4f}\")\n    print(f\"Mean Validation QWK: {np.mean(validation_scores):.4f}\")\n\n    # Optimize the threshold for rounding predictions\n    KappaOptimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    \n    # Apply optimized threshold to out-of-fold predictions\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tuned_kappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"Optimized QWK SCORE: {tuned_kappa:.3f}\")\n    print(\"-----------------------------------------\")\n\n    # Average test predictions and apply optimized threshold\n    average_test_preds = test_preds.mean(axis=1)\n    tuned_test_preds = threshold_Rounder(average_test_preds, KappaOptimizer.x)\n    \n    # Create submission DataFrame\n    submission = pd.DataFrame({\n        'id': sample['id'],  # Assuming 'sample' is defined with an 'id' column\n        'sii': tuned_test_preds\n    })\n    \n    # Clean up memory\n    gc.collect()\n    \n    return submission","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': -1,#9,\n    'num_leaves': 478,\n}\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 0,#6,\n    'n_estimators': 200,\n}\n\nCAT_Params = {\n    'iterations': 500,\n    'learning_rate': 0.1,\n    'depth': 6,\n}\n\n# Create models\nlgb_model = LGBMRegressor(**Params)\nxgb_model = XGBRegressor(**XGB_Params)\ncat_model = CatBoostRegressor(**CAT_Params, silent=True)\n\n# Define a voting regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lgb', lgb_model),\n    ('xgb', xgb_model),\n    ('cat', cat_model)\n])\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfrom IPython.display import clear_output\n# Train model\nsubmission = TrainML(voting_model, test)\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}