{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import packages","metadata":{"id":"iYUkPChKWWdy"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport re\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport time\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_selection import VarianceThreshold\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.ensemble import VotingClassifier, StackingRegressor, HistGradientBoostingRegressor, ExtraTreesRegressor, GradientBoostingRegressor\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.linear_model import LogisticRegression, ElasticNet, LinearRegression, HuberRegressor\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.svm import SVR","metadata":{"id":"ngv3qGRxWbSW","execution":{"iopub.status.busy":"2024-10-11T03:32:10.822918Z","iopub.execute_input":"2024-10-11T03:32:10.823380Z","iopub.status.idle":"2024-10-11T03:32:10.832965Z","shell.execute_reply.started":"2024-10-11T03:32:10.823338Z","shell.execute_reply":"2024-10-11T03:32:10.831446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%%capture\n\nfrom IPython.core.interactiveshell import InteractiveShell as IS; IS.ast_node_interactivity = \"all\"\nos.environ['TF_DETERMINISTIC_OPS'] = '1'; os.environ['TF_CUDNN_DETERMINISTIC'] = '1'; # allows seeding RNG on GPU\nToCSV = lambda df, fname: df.round(2).to_csv(f'{fname}.csv', index_label='id') # rounds values to 2 decimals\n\nclass Timer():\n  def __init__(self, lim:'RunTimeLimit'=60): self.t0, self.lim, _ = time.time(), lim, print(f'⏳ started. You have {lim} sec. Good luck!')\n  def ShowTime(self):\n    msg = f'Runtime is {time.time()-self.t0:.0f} sec'\n    print(f'\\033[91m\\033[1m' + msg + f' > {self.lim} sec limit!!!\\033[0m' if (time.time()-self.t0-1) > self.lim else msg)\n\nnp.set_printoptions(linewidth=100, precision=2, edgeitems=2, suppress=True)\npd.set_option('display.max_columns', 20, 'display.precision', 2, 'display.max_rows', 100)","metadata":{"id":"krJz4-p_efwt","outputId":"f18a1208-a73b-4b37-c5d1-567dd3fa5aa2","execution":{"iopub.status.busy":"2024-10-11T03:32:10.843088Z","iopub.execute_input":"2024-10-11T03:32:10.844432Z","iopub.status.idle":"2024-10-11T03:32:10.861543Z","shell.execute_reply.started":"2024-10-11T03:32:10.844378Z","shell.execute_reply":"2024-10-11T03:32:10.860327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Dataset","metadata":{"id":"mKKn8Lstd7t0"}},{"cell_type":"code","source":"def load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n\n    return df","metadata":{"id":"_Z62Np5i6sjD","execution":{"iopub.status.busy":"2024-10-11T03:32:10.863993Z","iopub.execute_input":"2024-10-11T03:32:10.864499Z","iopub.status.idle":"2024-10-11T03:32:10.873979Z","shell.execute_reply.started":"2024-10-11T03:32:10.864444Z","shell.execute_reply":"2024-10-11T03:32:10.872823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    # Try to load from Kaggle input directory first\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n    print(\"Loaded data from Kaggle input directory.\")\nexcept FileNotFoundError:\n    # If files not found in Kaggle input, load from local directory\n    train = pd.read_csv('train.csv')\n    test = pd.read_csv('test.csv')\n    sample = pd.read_csv('sample_submission.csv')\n    print(\"Loaded data from local directory.\")\n","metadata":{"id":"6oH4RFdO6b6M","outputId":"2608d401-9f27-466c-e0a2-09d78d6ea982","execution":{"iopub.status.busy":"2024-10-11T03:32:10.875625Z","iopub.execute_input":"2024-10-11T03:32:10.876098Z","iopub.status.idle":"2024-10-11T03:32:10.950531Z","shell.execute_reply.started":"2024-10-11T03:32:10.876056Z","shell.execute_reply":"2024-10-11T03:32:10.949271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(); print(f\"train shape: {train.shape}\")","metadata":{"id":"CGm7L2ZQjLOx","outputId":"7341fba9-b48f-410d-9d18-b2a2c8dd9785","execution":{"iopub.status.busy":"2024-10-11T03:32:10.952819Z","iopub.execute_input":"2024-10-11T03:32:10.953212Z","iopub.status.idle":"2024-10-11T03:32:10.983820Z","shell.execute_reply.started":"2024-10-11T03:32:10.953172Z","shell.execute_reply":"2024-10-11T03:32:10.982646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# null values in train\nprint(f\"Null values in train dataset {train.isnull().sum()}\")","metadata":{"id":"Sq86PY2EjQVa","outputId":"4f04fb95-21f7-410a-81b6-ad6a09afd89e","execution":{"iopub.status.busy":"2024-10-11T03:32:10.985541Z","iopub.execute_input":"2024-10-11T03:32:10.985903Z","iopub.status.idle":"2024-10-11T03:32:11.001406Z","shell.execute_reply.started":"2024-10-11T03:32:10.985857Z","shell.execute_reply":"2024-10-11T03:32:11.000206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = train.drop('sii', axis=1)\ntrain_y = train['sii']","metadata":{"id":"9Nz_EpENlJ1B","execution":{"iopub.status.busy":"2024-10-11T03:32:11.002727Z","iopub.execute_input":"2024-10-11T03:32:11.003135Z","iopub.status.idle":"2024-10-11T03:32:11.014734Z","shell.execute_reply.started":"2024-10-11T03:32:11.003095Z","shell.execute_reply":"2024-10-11T03:32:11.013678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace infinite values with NaN\n# df.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# # Impute NaN with the median of each column\n# imputer = SimpleImputer(strategy='median')\n# df_imputed = pd.DataFrame(imputer.fit_transform(df), columns=df.columns)","metadata":{"id":"RIWiqJogt9pF","execution":{"iopub.status.busy":"2024-10-11T03:32:11.015950Z","iopub.execute_input":"2024-10-11T03:32:11.016313Z","iopub.status.idle":"2024-10-11T03:32:11.025797Z","shell.execute_reply.started":"2024-10-11T03:32:11.016275Z","shell.execute_reply":"2024-10-11T03:32:11.024502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Boxplots for outliers\ndef plot_boxplots_in_batches(df, batch_size=10):\n    numeric_columns = df.select_dtypes(include=[np.number]).columns\n    total_columns = len(numeric_columns)\n\n    for start in range(0, total_columns, batch_size):\n        end = min(start + batch_size, total_columns)\n        batch_columns = numeric_columns[start:end]\n\n        nrows = int(np.ceil(len(batch_columns) / 3))\n        fig, axes = plt.subplots(nrows=nrows, ncols=3, figsize=(16, nrows * 4))\n        axes = axes.flatten()\n\n        for i, col in enumerate(batch_columns):\n            sns.boxplot(x=df[col], ax=axes[i], color='orange')\n            axes[i].set_title(f'Boxplot of {col}')\n            axes[i].set_xlabel(col)\n\n        for j in range(i + 1, len(axes)):\n            fig.delaxes(axes[j])\n\n        plt.tight_layout()\n        plt.show()","metadata":{"id":"0bqs3okA1gio","execution":{"iopub.status.busy":"2024-10-11T03:32:11.029729Z","iopub.execute_input":"2024-10-11T03:32:11.030158Z","iopub.status.idle":"2024-10-11T03:32:11.040991Z","shell.execute_reply.started":"2024-10-11T03:32:11.030114Z","shell.execute_reply":"2024-10-11T03:32:11.039772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation heat map\ndef plot_correlation_heatmap(df):\n  # Select only numeric columns for correlation calculation\n  numeric_df = df.select_dtypes(include=['number'])\n\n  plt.figure(figsize=(12, 10))\n  corr_matrix = numeric_df.corr()\n  sns.heatmap(corr_matrix, annot=False, cmap='coolwarm', vmin=-1, vmax=1)\n  plt.title('Correlation Heatmap')\n  plt.show()","metadata":{"id":"WMvSfhaD1u_6","execution":{"iopub.status.busy":"2024-10-11T03:32:11.042531Z","iopub.execute_input":"2024-10-11T03:32:11.042995Z","iopub.status.idle":"2024-10-11T03:32:11.056438Z","shell.execute_reply.started":"2024-10-11T03:32:11.042941Z","shell.execute_reply":"2024-10-11T03:32:11.055374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handle missing data by using mean/median/mode imputation\ndef handle_missing_data(df, strategy='mean'):\n    for col in df.columns:\n        if df[col].isnull().sum() > 0:\n            if df[col].dtype == 'object' or df[col].dtype =='category':\n                # For categorical data, use mode imputation\n                df[col].fillna(df[col].mode()[0], inplace=True)\n            else:\n                # For numerical data, use the specified strategy\n                if strategy == 'mean':\n                    df[col].fillna(df[col].mean(), inplace=True)\n                elif strategy == 'median':\n                    df[col].fillna(df[col].median(), inplace=True)\n                elif strategy == 'mode':\n                    df[col].fillna(df[col].mode()[0], inplace=True)\n    return df","metadata":{"id":"y3MIAIAT18xj","execution":{"iopub.status.busy":"2024-10-11T03:32:11.058011Z","iopub.execute_input":"2024-10-11T03:32:11.058610Z","iopub.status.idle":"2024-10-11T03:32:11.069770Z","shell.execute_reply.started":"2024-10-11T03:32:11.058554Z","shell.execute_reply":"2024-10-11T03:32:11.068568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_x)","metadata":{"id":"btLhUBA7sqZ2","outputId":"70ce78df-4f16-498f-fd81-4d58ed7683f9","execution":{"iopub.status.busy":"2024-10-11T03:32:11.073457Z","iopub.execute_input":"2024-10-11T03:32:11.073924Z","iopub.status.idle":"2024-10-11T03:32:11.083642Z","shell.execute_reply.started":"2024-10-11T03:32:11.073881Z","shell.execute_reply":"2024-10-11T03:32:11.082414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x_cleaned = handle_missing_data(train_x, strategy='mean')","metadata":{"id":"_BunoRwRODOM","execution":{"iopub.status.busy":"2024-10-11T03:32:11.085676Z","iopub.execute_input":"2024-10-11T03:32:11.086186Z","iopub.status.idle":"2024-10-11T03:32:11.157930Z","shell.execute_reply.started":"2024-10-11T03:32:11.086124Z","shell.execute_reply":"2024-10-11T03:32:11.156564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x_cleaned.describe()","metadata":{"id":"pKP0ujHwOPzW","outputId":"5f6e90af-3316-4d88-cb71-477480a89026","execution":{"iopub.status.busy":"2024-10-11T03:32:11.159679Z","iopub.execute_input":"2024-10-11T03:32:11.160163Z","iopub.status.idle":"2024-10-11T03:32:11.335263Z","shell.execute_reply.started":"2024-10-11T03:32:11.160108Z","shell.execute_reply":"2024-10-11T03:32:11.334022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import  warnings\nwarnings.filterwarnings('ignore')\n# plot_boxplots_in_batches(train_x, batch_size=10)","metadata":{"id":"3marU_n7FbOI","outputId":"84e348f7-8f2b-40b2-f2e4-07ee2e2c4b2b","execution":{"iopub.status.busy":"2024-10-11T03:32:11.336930Z","iopub.execute_input":"2024-10-11T03:32:11.337448Z","iopub.status.idle":"2024-10-11T03:32:11.342783Z","shell.execute_reply.started":"2024-10-11T03:32:11.337397Z","shell.execute_reply":"2024-10-11T03:32:11.341680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_correlation_heatmap(train_x_cleaned)","metadata":{"id":"OrE7tCcn_d1-","outputId":"25ebdecf-6bf0-458d-cd00-8de8cd2f10d5","execution":{"iopub.status.busy":"2024-10-11T03:32:11.343902Z","iopub.execute_input":"2024-10-11T03:32:11.344254Z","iopub.status.idle":"2024-10-11T03:32:12.492511Z","shell.execute_reply.started":"2024-10-11T03:32:11.344217Z","shell.execute_reply":"2024-10-11T03:32:12.491173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_feature_types(df):\n    # Get the list of categorical and numerical features\n    categorical_features = df.select_dtypes(include=['object']).columns.tolist()\n    numerical_features = df.select_dtypes(include=['number']).columns.tolist()\n\n    return categorical_features, numerical_features","metadata":{"id":"YfjVcxr_Tcns","execution":{"iopub.status.busy":"2024-10-11T03:32:12.498582Z","iopub.execute_input":"2024-10-11T03:32:12.499056Z","iopub.status.idle":"2024-10-11T03:32:12.506532Z","shell.execute_reply.started":"2024-10-11T03:32:12.499005Z","shell.execute_reply":"2024-10-11T03:32:12.505201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features, numerical_features = get_feature_types(train_x_cleaned)\nprint(\"Categorical Features:\", categorical_features)\nprint(\"Numerical Features:\", numerical_features)","metadata":{"id":"PrVb4TetUGX-","outputId":"002200f4-9bf4-422f-ae0a-96f5e5e8cc81","execution":{"iopub.status.busy":"2024-10-11T03:32:12.507977Z","iopub.execute_input":"2024-10-11T03:32:12.508359Z","iopub.status.idle":"2024-10-11T03:32:12.520184Z","shell.execute_reply.started":"2024-10-11T03:32:12.508306Z","shell.execute_reply":"2024-10-11T03:32:12.518926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot to visualize distribution of numerical features\ndef plot_numerical_distributions(df, numerical_features):\n    for col in numerical_features:\n        plt.figure(figsize=(8, 6))\n        sns.histplot(df[col], kde=True)\n        plt.title(f'Distribution of {col}')\n        plt.xlabel(col)\n        plt.ylabel('Frequency')\n        plt.show()","metadata":{"id":"Fd-RkVFQwv83","execution":{"iopub.status.busy":"2024-10-11T03:32:12.521818Z","iopub.execute_input":"2024-10-11T03:32:12.522259Z","iopub.status.idle":"2024-10-11T03:32:12.529861Z","shell.execute_reply.started":"2024-10-11T03:32:12.522217Z","shell.execute_reply":"2024-10-11T03:32:12.528462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_numerical_distributions(train_x_cleaned, numerical_features=numerical_features)","metadata":{"id":"ZLJXKk5gxrFn","outputId":"e61c65a1-2cce-4842-dfb6-9cfb64378277","execution":{"iopub.status.busy":"2024-10-11T03:32:12.531512Z","iopub.execute_input":"2024-10-11T03:32:12.532059Z","iopub.status.idle":"2024-10-11T03:32:12.541238Z","shell.execute_reply.started":"2024-10-11T03:32:12.532007Z","shell.execute_reply":"2024-10-11T03:32:12.539936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_top_outliers(df, num_features=5, numerical_features=None):\n    if numerical_features is None:\n      numerical_features = df.select_dtypes(include=['number']).columns.tolist()\n    else:\n      numerical_features = numerical_features\n\n    outlier_counts = {}\n\n    # Calculate outliers for each numerical feature using IQR\n    for col in numerical_features:\n        Q1 = df[col].quantile(0.25)\n        Q3 = df[col].quantile(0.75)\n        IQR = Q3 - Q1\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        # Count the number of outliers\n        outliers = ((df[col] < lower_bound) | (df[col] > upper_bound)).sum()\n        outlier_counts[col] = outliers\n\n    # Sort features by number of outliers\n    sorted_outliers = sorted(outlier_counts.items(), key=lambda x: x[1], reverse=True)\n\n    # Select the top features\n    top_features = sorted_outliers[:num_features]\n\n    # Convert to DataFrame for plotting\n    outlier_df = pd.DataFrame(top_features, columns=['Feature', 'Outlier Count'])\n\n    # Plot the top features with the most outliers\n    plt.figure(figsize=(10, 6))\n    sns.barplot(x='Feature', y='Outlier Count', data=outlier_df, palette='Blues')\n    plt.title(f'Top {num_features} Features with Most Outliers')\n    plt.xlabel('Feature')\n    plt.ylabel('Outlier Count')\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"id":"95-Zb8gS_aRO","execution":{"iopub.status.busy":"2024-10-11T03:32:12.543643Z","iopub.execute_input":"2024-10-11T03:32:12.544113Z","iopub.status.idle":"2024-10-11T03:32:12.558079Z","shell.execute_reply.started":"2024-10-11T03:32:12.544070Z","shell.execute_reply":"2024-10-11T03:32:12.556439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_outliers(train_x_cleaned, num_features=10, numerical_features=numerical_features)","metadata":{"id":"PO9a2JJhAe6t","outputId":"7cef276c-83ae-480b-c207-346a24813ad8","execution":{"iopub.status.busy":"2024-10-11T03:32:12.559675Z","iopub.execute_input":"2024-10-11T03:32:12.560660Z","iopub.status.idle":"2024-10-11T03:32:13.074727Z","shell.execute_reply.started":"2024-10-11T03:32:12.560616Z","shell.execute_reply":"2024-10-11T03:32:13.073433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_feature_vs_target(df, feature, target='sii'):\n    # Boxplot for the specified feature against the target variable\n    plt.figure(figsize=(8, 6))\n    sns.boxplot(data=df, x=df[target], y=df[feature])\n    plt.title(f'{target} vs {feature}')\n    plt.xlabel(target)\n    plt.ylabel(feature)\n    plt.show()","metadata":{"id":"_-oSU-sPFpfE","execution":{"iopub.status.busy":"2024-10-11T03:32:13.076498Z","iopub.execute_input":"2024-10-11T03:32:13.076963Z","iopub.status.idle":"2024-10-11T03:32:13.083851Z","shell.execute_reply.started":"2024-10-11T03:32:13.076920Z","shell.execute_reply":"2024-10-11T03:32:13.082771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_vs_target(train, feature='BIA-BIA_Fat')","metadata":{"id":"_f5CcSGcFw9s","outputId":"619692af-c179-4669-aa3e-5168ea340293","execution":{"iopub.status.busy":"2024-10-11T03:32:13.085551Z","iopub.execute_input":"2024-10-11T03:32:13.085987Z","iopub.status.idle":"2024-10-11T03:32:13.387643Z","shell.execute_reply.started":"2024-10-11T03:32:13.085945Z","shell.execute_reply":"2024-10-11T03:32:13.386423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test, id_train, id_test = train_test_split(train_x_cleaned, train_y, train['id'], test_size=0.2, random_state=42)\n\nlow_variance_filter = VarianceThreshold(threshold=0.05)\n\n# Define preprocessing steps\nnumeric_features = train_x.select_dtypes(include=['int64', 'float64']).columns\ncategorical_features = train_x.select_dtypes(include=['object']).columns\n\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ])\n\n# Define the models\nxgb = XGBClassifier(use_label_encoder=False, eval_metric='logloss')\nlgbm = LGBMClassifier(\n    num_leaves=31,\n    max_depth=5,\n    min_data_in_leaf=20,\n    learning_rate=0.05,\n    n_estimators=200\n)\n\n# Ensemble model with Voting Classifier\nensemble = VotingClassifier(estimators=[\n    ('xgb', xgb), \n    ('lgbm', lgbm)\n], voting='soft')\n\n# Create the pipeline\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('variance_filter', low_variance_filter),  # Add variance threshold step\n    ('classifier', ensemble)\n])","metadata":{"id":"LBIX4c7B8m4_","execution":{"iopub.status.busy":"2024-10-11T03:32:13.389032Z","iopub.execute_input":"2024-10-11T03:32:13.389419Z","iopub.status.idle":"2024-10-11T03:32:13.414158Z","shell.execute_reply.started":"2024-10-11T03:32:13.389380Z","shell.execute_reply":"2024-10-11T03:32:13.413010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the pipeline\nX_train_vals = X_train[y_train.notna()]\ny_train_vals = y_train[y_train.notna()]\npipeline.fit(X_train_vals, y_train_vals)","metadata":{"id":"EZKU7pfR9O_R","outputId":"72aeb8ae-5582-42a1-ae89-c62a8c74f22e","execution":{"iopub.status.busy":"2024-10-11T03:32:13.415602Z","iopub.execute_input":"2024-10-11T03:32:13.416776Z","iopub.status.idle":"2024-10-11T03:32:15.870416Z","shell.execute_reply.started":"2024-10-11T03:32:13.416705Z","shell.execute_reply":"2024-10-11T03:32:15.869127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict on the test data\ny_pred = pipeline.predict(X_test)","metadata":{"id":"uz4639M49Pua","execution":{"iopub.status.busy":"2024-10-11T03:32:15.871833Z","iopub.execute_input":"2024-10-11T03:32:15.872195Z","iopub.status.idle":"2024-10-11T03:32:15.933158Z","shell.execute_reply.started":"2024-10-11T03:32:15.872157Z","shell.execute_reply":"2024-10-11T03:32:15.931817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert predictions into a DataFrame for CSV output\nsubmission_df = pd.DataFrame({\n    'id': X_test['id'],  # Assuming 'id' is available as the index or part of the features\n    'sii': y_pred\n})","metadata":{"id":"to8T3HSY-IgN","execution":{"iopub.status.busy":"2024-10-11T03:32:15.934444Z","iopub.execute_input":"2024-10-11T03:32:15.934791Z","iopub.status.idle":"2024-10-11T03:32:15.941083Z","shell.execute_reply.started":"2024-10-11T03:32:15.934755Z","shell.execute_reply":"2024-10-11T03:32:15.939746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='most_frequent') #Choose most frequent for categorical target\ny_test_imputed = imputer.fit_transform(y_test.values.reshape(-1, 1))\ny_test_imputed = y_test_imputed.ravel()\n\naccuracy = accuracy_score(y_test_imputed, y_pred)\nprint(f'Ensemble Model Accuracy: {accuracy:.4f}')","metadata":{"id":"maPDIYU39THx","outputId":"19195bfc-6f65-4c4e-8789-f718aa231a12","execution":{"iopub.status.busy":"2024-10-11T03:32:15.942772Z","iopub.execute_input":"2024-10-11T03:32:15.943231Z","iopub.status.idle":"2024-10-11T03:32:15.958115Z","shell.execute_reply.started":"2024-10-11T03:32:15.943189Z","shell.execute_reply":"2024-10-11T03:32:15.956852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the output to CSV\nToCSV(submission_df, 'submission')\n\n# Evaluate the model accuracy\n# accuracy = accuracy_score(y_test, y_pred)\n# print(f'Ensemble Model Accuracy: {accuracy:.4f}')","metadata":{"id":"2dtXML319Joq","execution":{"iopub.status.busy":"2024-10-11T03:32:15.959741Z","iopub.execute_input":"2024-10-11T03:32:15.960217Z","iopub.status.idle":"2024-10-11T03:32:15.973873Z","shell.execute_reply.started":"2024-10-11T03:32:15.960152Z","shell.execute_reply":"2024-10-11T03:32:15.972583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Trial 1","metadata":{"id":"rF_IaEhXbmvJ"}},{"cell_type":"code","source":"# # Create an imputer for missing values\n# knn_imputer = KNNImputer(n_neighbors=5)\n# # Model parameters for LightGBM\n\n# Params = {'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207, 'min_data_in_leaf': 41,\n#                 'feature_fraction': 0.6385678848225935, 'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6,\n#                             'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983}\n\n#     # XGBoost parameters\n# XGB_Params = {'learning_rate': 0.007356059931165658, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\n\n\n# CatBoost_Params = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}\n\n#     # Create model instances\n# Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=800)\n# XGB_Model = XGBRegressor(**XGB_Params)\n# CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n# MLP = make_pipeline(knn_imputer,MLPRegressor(hidden_layer_sizes=(100,), max_iter=100))\n# SVR_Model = make_pipeline(knn_imputer,SVR(kernel='rbf', C=1.0, epsilon=0.1))\n# KNN = make_pipeline(knn_imputer,KNeighborsRegressor(n_neighbors=5))\n# Huber =  make_pipeline(knn_imputer,HuberRegressor())\n# meta_model = make_pipeline(knn_imputer, GradientBoostingRegressor(n_estimators=100, learning_rate=0.05))\n\n# # Stacking Regressor\n# stacking_model = StackingRegressor(\n#     estimators=[\n#         ('lightgbm', Light),\n#         ('xgboost', XGB_Model),\n#         ('catboost', CatBoost_Model),\n\n\n#     ],\n#     final_estimator=meta_model,  # Meta-model\n#     passthrough=True  # Option to pass original features along with predictions from base models\n# )\n\n# # Train the Stacking Regressor\n# Submission = TrainML(stacking_model, test)\n\n# # Save submission\n# Submission.to_csv('submission.csv', index=False)\n# print(Submission['sii'].value_counts())","metadata":{"id":"4SkObvXZjAL4","execution":{"iopub.status.busy":"2024-10-11T03:32:15.975436Z","iopub.execute_input":"2024-10-11T03:32:15.975891Z","iopub.status.idle":"2024-10-11T03:32:15.987382Z","shell.execute_reply.started":"2024-10-11T03:32:15.975839Z","shell.execute_reply":"2024-10-11T03:32:15.986333Z"},"trusted":true},"execution_count":null,"outputs":[]}]}