{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"text-align:center\"><span style=\"background-color:#a8edc4;color:black;padding:10px;border-radius:40px;\">💻 Blend XGB, LGBM, CAT - EDA 📊</span></div>\n\n\n![](https://i.postimg.cc/HL1fFP5X/pexels-ron-lach-9783375.jpg)\n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🎒Import Libraries</span>\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.ticker import MaxNLocator\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\n# pd.options.display.max_columns = None\n\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import RobustScaler\nfrom mlxtend.regressor import StackingCVRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nimport lightgbm as lgb\n\nfrom datetime import datetime\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import PowerTransformer\nimport scipy.stats as stats\n\nfrom sklearn.decomposition import SparsePCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer \n\nn_splits = 6","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-26T11:36:50.498644Z","iopub.execute_input":"2024-11-26T11:36:50.499057Z","iopub.status.idle":"2024-11-26T11:36:55.953011Z","shell.execute_reply.started":"2024-11-26T11:36:50.499018Z","shell.execute_reply":"2024-11-26T11:36:55.951967Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑Explaining Python Libraries</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Essential Libraries</h2>\n  <ul>\n    <li><strong>NumPy:</strong> Foundation for numerical computations</li>\n    <li><strong>Pandas:</strong> Data structures and analysis tools</li>\n    <li><strong>Polars:</strong> High-performance data processing library</li>\n    <li><strong>Scikit-learn:</strong> Machine learning algorithms and tools</li>\n    <li><strong>SciPy:</strong> Scientific computing library</li>\n  </ul>\n  <h2 style=\"color:black;\">Operating System and Visualization Tools</h2>\n  <ul>\n    <li><strong>OS:</strong> Interacts with the operating system</li>\n    <li><strong>Matplotlib:</strong> Plotting library for creating visualizations</li>\n    <li><strong>Seaborn:</strong> High-level interface for drawing statistical graphics</li>\n    <li><strong>Matplotlib.ticker:</strong> Customizes tick labels and formatting</li>\n  </ul>\n  <h2 style=\"color:black;\">Text Processing and Output Formatting</h2>\n  <ul>\n    <li><strong>RE:</strong> Regular expression operations</li>\n    <li><strong>Colorama:</strong> Colored text output in the terminal</li>\n    <li><strong>TQDM:</strong> Progress bar for long-running operations</li>\n    <li><strong>IPython.display:</strong> Displays output in Jupyter notebooks</li>\n    <li><strong>Concurrent.futures:</strong> Parallel execution of tasks</li>\n    <li><strong>Warnings:</strong> Manages warning messages</li>\n  </ul>\n  <h2 style=\"color:black;\">Machine Learning Models and Tools</h2>\n  <ul>\n    <li><strong>Scikit-learn.model_selection:</strong> Data splitting and model evaluation</li>\n    <li><strong>Scikit-learn.metrics:</strong> Calculation of performance metrics</li>\n    <li><strong>Scikit-learn.pipeline:</strong> Chaining multiple data processing steps</li>\n    <li><strong>Scikit-learn.preprocessing:</strong> Data normalization and scaling</li>\n    <li><strong>mlxtend.regressor.StackingCVRegressor:</strong> Stacking ensemble method</li>\n    <li><strong>LightGBM:</strong> Gradient boosting machine</li>\n    <li><strong>CatBoost:</strong> Gradient boosting machine</li>\n    <li><strong>XGBoost:</strong> Gradient boosting machine</li>\n  </ul>\n  <h2 style=\"color:black;\">Additional Modules and Settings</h2>\n  <ul>\n    <li><strong>Datetime:</strong> Works with dates and times</li>\n    <li><strong>Scikit-learn.model_selection.KFold:</strong> k-fold cross-validation</li>\n    <li><strong>Scikit-learn.preprocessing.LabelEncoder:</strong> Encodes categorical variables</li>\n    <li><strong>Hyperparameter:</strong> `n_splits = 6`</li>\n    <li><strong>Configuration:</strong> `warnings.filterwarnings('ignore')` and `pd.options.display.max_columns = None`</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#79a5ed;padding:10px;border-radius:40px;\">✨Preprocessing</span>","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df = df[df['non-wear_flag'] != 1]\n    df.drop('step', axis=1, inplace=True)\n    df.drop('time_of_day', axis=1, inplace=True)\n    df.drop('weekday', axis=1, inplace=True)\n    df.drop('quarter', axis=1, inplace=True)\n    df.drop('relative_date_PCIAT', axis=1, inplace=True)\n    \n    # Calculate the mean for each column\n#     means = df.mean().values    \n#     return means, filename.split('=')[1]\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\"\"\"This Mapping Works Fine For me I also Check Each Values in Train and test Using Logic. There no Data Lekage.\"\"\"\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:36:55.955473Z","iopub.execute_input":"2024-11-26T11:36:55.956188Z","iopub.status.idle":"2024-11-26T11:38:10.395205Z","shell.execute_reply.started":"2024-11-26T11:36:55.956136Z","shell.execute_reply":"2024-11-26T11:38:10.394102Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑Explaining Preprocessing</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Time Series Loading Function (<strong>load_time_series</strong>)</h2>\n  <ul>\n    <li>Takes a directory path containing time series data in Parquet format as input.</li>\n    <li>Uses <strong>os.listdir</strong> to get a list of file names in the directory.</li>\n    <li>Utilizes a <strong>ThreadPoolExecutor</strong> to parallelize the processing of each file.</li>\n    <li>The <strong>process_file</strong> function is called for each file.</li>\n    <li><strong>process_file</strong> reads the Parquet file using <strong>pd.read_parquet</strong>.</li>\n    <li>It drops the \"step\" column (assuming it's not relevant).</li>\n    <li>It calculates descriptive statistics for the remaining columns using <strong>df.describe().values.reshape(-1)</strong>.</li>\n    <li>The function returns a tuple containing the statistics and the file name (split to extract the ID).</li>\n    <li>Finally, the function combines the statistics from all files into a DataFrame and sets the \"id\" column based on the file names.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Data Loading and Merging</h2>\n  <ul>\n    <li>Three datasets are loaded: <strong>train.csv</strong>, <strong>test.csv</strong>, and <strong>sample_submission.csv</strong>.</li>\n    <li>The <strong>load_time_series</strong> function is used to load time series data for both training and testing sets.</li>\n    <li>The time series dataframes are merged with the main datasets based on the \"id\" column.</li>\n    <li>The \"id\" column is dropped from the main datasets after merging.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Feature Selection and Pre-processing</h2>\n  <ul>\n    <li>A list of pre-defined features (<strong>featuresCols</strong>) is created.</li>\n    <li>This list combines features from the main datasets and the time series statistics.</li>\n    <li>Both training and testing sets are filtered to include only these features.</li>\n    <li>Missing values in the \"sii\" column (presumably the target variable) are dropped from the training set.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Categorical Feature Handling</h2>\n  <ul>\n    <li>A list of categorical features (<strong>cat_c</strong>) is defined.</li>\n    <li>A function <strong>update</strong> is defined to fill missing values in these features with \"Missing\" and convert them to categorical data type.</li>\n    <li>Both training and testing sets are updated using this function.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Feature Encoding</h2>\n  <ul>\n    <li>A function <strong>create_mapping</strong> creates a dictionary mapping unique values in a categorical column to integer values.</li>\n    <li>Separate mappings are created for training and testing sets to avoid data leakage (using the training set information to predict the test set).</li>\n    <li>The categorical features are replaced with their corresponding integer values based on the mappings.</li>\n    <li>This step encodes the categorical information numerically for the machine learning model.</li>\n  </ul>\n\n  <h2 style=\"color:black;\">Printing Final Shapes</h2>\n  <ul>\n    <li>The function prints the final shapes of the training and testing sets after all processing.</li>\n  </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🔎View Data</span>","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:10.396486Z","iopub.execute_input":"2024-11-26T11:38:10.396806Z","iopub.status.idle":"2024-11-26T11:38:10.426583Z","shell.execute_reply.started":"2024-11-26T11:38:10.396773Z","shell.execute_reply":"2024-11-26T11:38:10.425370Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:10.428701Z","iopub.execute_input":"2024-11-26T11:38:10.429788Z","iopub.status.idle":"2024-11-26T11:38:10.453774Z","shell.execute_reply.started":"2024-11-26T11:38:10.429745Z","shell.execute_reply":"2024-11-26T11:38:10.452612Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet')\nactigraphy","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:10.455359Z","iopub.execute_input":"2024-11-26T11:38:10.455737Z","iopub.status.idle":"2024-11-26T11:38:10.622105Z","shell.execute_reply.started":"2024-11-26T11:38:10.455697Z","shell.execute_reply":"2024-11-26T11:38:10.620880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📊 Exploratory Data Analysis</span>","metadata":{}},{"cell_type":"code","source":"vc = train['Basic_Demos-Enroll_Season'].value_counts()\n\n# Map labels to seasons\nseason_map = {0: 'Spring', 1: 'Summer', 2: 'Fall', 3: 'Winter'}\n\n# Create labels using the map\nlabels = [season_map[label] for label in vc.index]\n\n# Plot the pie chart with the updated labels\nplt.pie(vc, labels=labels)\n# Plot the pie chart\nplt.pie(vc.values, labels=labels, autopct=\"%1.1f%%\")  # Add percentages to pie slices\nplt.title('Season of Enrollment')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:10.623360Z","iopub.execute_input":"2024-11-26T11:38:10.623683Z","iopub.status.idle":"2024-11-26T11:38:10.846547Z","shell.execute_reply.started":"2024-11-26T11:38:10.623651Z","shell.execute_reply":"2024-11-26T11:38:10.844646Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It seems we have almost similar enrollment in all seasons","metadata":{}},{"cell_type":"code","source":"# Assuming you have your DataFrame 'train' with the column 'PreInt_EduHx-computerinternet_hoursday'\n\nvc = train['PreInt_EduHx-computerinternet_hoursday'].value_counts()  # Count occurrences of each value\n\n# Improved InternetHours_map with clearer categories\nInternetHours_map = {\n    0: 'Less than 1 hour',\n    1: 'Around 1 hour',\n    2: 'Around 2 hours',\n    3: 'More than 3 hours'\n}\n\n# Create labels using the map\nlabels = [InternetHours_map[label] for label in vc.index]\n\n# Create a bar chart\nplt.bar(labels, vc.values)  # Use labels for bar positions\n\n# Customize the bar chart for better visualization\nplt.xlabel('Hours Spent on Internet per Day')\nplt.ylabel('Number of People')\nplt.title('Distribution of Daily Internet Usage')\nplt.xticks(rotation=45, ha='right')  # Rotate category labels for readability\nplt.tight_layout()  # Adjust spacing to prevent overlapping elements\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:10.848860Z","iopub.execute_input":"2024-11-26T11:38:10.849481Z","iopub.status.idle":"2024-11-26T11:38:11.080491Z","shell.execute_reply.started":"2024-11-26T11:38:10.849419Z","shell.execute_reply":"2024-11-26T11:38:11.079280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In our dataset majority of people use internet for less than 1 hour. It can be seen that people using internet for more than 3 hours are very less. ","metadata":{}},{"cell_type":"code","source":"vc = train['sii'].value_counts()\n\n# Map labels to sii\nsii_map = {0: 'None', 1: 'Mild', 2: 'Moderate', 3: 'Severe'}\n\n# Create labels using the map\nlabels = [sii_map[label] for label in vc.index]\n\n# Plot the pie chart with the updated labels\nplt.pie(vc, labels=labels)\n# Plot the pie chart\nplt.pie(vc.values, labels=labels, autopct=\"%1.1f%%\")  # Add percentages to pie slices\nplt.title('Target Variable sii')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:11.081957Z","iopub.execute_input":"2024-11-26T11:38:11.082326Z","iopub.status.idle":"2024-11-26T11:38:11.278452Z","shell.execute_reply.started":"2024-11-26T11:38:11.082289Z","shell.execute_reply":"2024-11-26T11:38:11.276913Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Pie chart makes it clear that majority of people i.e 58.3% don't have any problem. While 26.7% have only mild problems. So, we can say that 85% of people in our data either have no or mild problem. While in the remaning 15% we see that 13.8% have moderate problem and only 1.2% people show severe problem. As a result in our dataset the number of people who actually have severe problem are only 1.2%, which is a very small percentage as compared to other 3 groups of people.","metadata":{}},{"cell_type":"code","source":"vc = train['PreInt_EduHx-computerinternet_hoursday'].value_counts()\n\n# Map labels for Internet Hours\nInternetHours_map = {0: 'Less than 1h/day', 1: 'Around 1h/day', 2: 'Around 2hs/day', 3: 'More than 3hs/day'}\n\n# Create labels using the map\nlabels = [InternetHours_map[label] for label in vc.index]\n\n# Plot the pie chart with the updated labels\nplt.pie(vc, labels=labels)\n# Plot the pie chart\nplt.pie(vc.values, labels=labels, autopct=\"%1.1f%%\")  # Add percentages to pie slices\nplt.title('Hours spent on internet per day')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:11.280397Z","iopub.execute_input":"2024-11-26T11:38:11.281113Z","iopub.status.idle":"2024-11-26T11:38:11.491840Z","shell.execute_reply.started":"2024-11-26T11:38:11.281035Z","shell.execute_reply":"2024-11-26T11:38:11.490365Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Viewing the pie chart above it is clear that the people who use internet excessively are only 9.9%. While all other users either use the internet for around 2 hours or less than it.","metadata":{}},{"cell_type":"code","source":"# Create the scatter plot\nplt.figure(figsize=(8, 6))  # Set the figure size for better visualization\nplt.scatter(train['sii'], train['PreInt_EduHx-computerinternet_hoursday'], color='royalblue', alpha=0.7)  # Adjust color and transparency\n\n# Add labels and title\nplt.xlabel('SII')\nplt.ylabel('PreInt_EduHx-computerinternet_hoursday')\nplt.title('Scatter Plot of SII vs. PreInt_EduHx-computerinternet_hoursday')\n\n# Add grid lines for better readability\nplt.grid(True)\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:11.498974Z","iopub.execute_input":"2024-11-26T11:38:11.499599Z","iopub.status.idle":"2024-11-26T11:38:11.806000Z","shell.execute_reply.started":"2024-11-26T11:38:11.499530Z","shell.execute_reply":"2024-11-26T11:38:11.804856Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here we have a scatter plot for our target variable i.e SII and number of hours for which internet was used. This plot is interesting but also strange. It shows that we have all possible SII values for all possible internet usage hours per day. This makes it difficult to distinguish and understand that how the SII value changes with changes in time of internet usage. A possible solution to better understand it will be to introduce count of computer/internet users along with SII and internet usage hours per day.","metadata":{}},{"cell_type":"code","source":"#https://www.kaggle.com/competitions/child-mind-institute-problematic-internet-use/discussion/546266\n\ntmp_train = train.copy()\ntmp = train[['sii','PreInt_EduHx-computerinternet_hoursday']].value_counts().reset_index()\ntmp[['sii','PreInt_EduHx-computerinternet_hoursday']] = tmp[['sii','PreInt_EduHx-computerinternet_hoursday']].astype(int)\n\ntmp = tmp.pivot(\n    columns=\"PreInt_EduHx-computerinternet_hoursday\",\n    index=\"sii\",\n    values=\"count\"\n).fillna(0)\n\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(tmp, annot=True, fmt=\".0f\", cmap=\"Blues\", cbar=True, linewidths=0.01, linecolor='lightgray')\n\nplt.xlabel('SII')\nplt.ylabel('Computer/Internet Hours per Day')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:11.807258Z","iopub.execute_input":"2024-11-26T11:38:11.807593Z","iopub.status.idle":"2024-11-26T11:38:12.122139Z","shell.execute_reply.started":"2024-11-26T11:38:11.807556Z","shell.execute_reply":"2024-11-26T11:38:12.120923Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In the above chart we have sii, PreInt_EduHx-computerinternet_hoursday and count of PreInt_EduHx-computerinternet_hoursday. These three colletively help in better understanding the data as compared to using only two features in the previous chart.","metadata":{}},{"cell_type":"code","source":"# Get value counts for gender\nvc = train['Basic_Demos-Sex'].value_counts()\n\n# Extract counts (assuming 'count' column doesn't exist)\ncounts = vc.values\n\n# Create labels (assuming 'count' column doesn't exist)\nlabels = ['Boys', 'Girls']  # Assuming 'Male' maps to 'Boys' and 'Female' to 'Girls'\n\n# Plot the pie chart\nplt.pie(counts, labels=labels, autopct=\"%1.1f%%\")  # Add percentages to pie slices\nplt.title('Gender')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:12.123563Z","iopub.execute_input":"2024-11-26T11:38:12.123966Z","iopub.status.idle":"2024-11-26T11:38:12.251289Z","shell.execute_reply.started":"2024-11-26T11:38:12.123929Z","shell.execute_reply":"2024-11-26T11:38:12.249793Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Majority of people are male in this dataset while we do have girls in a good number too.","metadata":{}},{"cell_type":"code","source":"#'train' is DataFrame\n_, axs = plt.subplots(2, 1, sharex=True)\n\nfor sex in range(2):\n  ax = axs.ravel()[sex]\n\n  # Filter by sex using boolean indexing\n  sex_filter = train['Basic_Demos-Sex'] == sex\n  vc = train[sex_filter]['Basic_Demos-Age'].value_counts()  # Access column after filtering\n\n  # Plot the bar chart\n  ax.bar(vc.index, vc.values, color=['lightblue', 'coral'][sex], label=['boys', 'girls'][sex])\n  ax.xaxis.set_major_locator(MaxNLocator(integer=True))\n  ax.set_ylabel('count')\n  ax.legend()\n\nplt.suptitle('Age distribution')\naxs.ravel()[1].set_xlabel('years')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:12.253092Z","iopub.execute_input":"2024-11-26T11:38:12.254136Z","iopub.status.idle":"2024-11-26T11:38:12.749789Z","shell.execute_reply.started":"2024-11-26T11:38:12.254059Z","shell.execute_reply":"2024-11-26T11:38:12.748655Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Majority of people in our dataset are young children or teenagers. We don't have any elder person. Remember we are being asked to determine SII for people in future.","metadata":{}},{"cell_type":"code","source":"# Create the scatter plot\nplt.figure(figsize=(8, 6))  # Set the figure size for better visualization\nplt.scatter(train['sii'], train['SDS-SDS_Total_Raw'], color='royalblue', alpha=0.7)  # Adjust color and transparency\n\n# Add labels and title\nplt.xlabel('SII')\nplt.ylabel('Sleep Disturbance Scale')\nplt.title('Scatter Plot of SII vs. Sleep Disturbance Scale')\n\n# Add grid lines for better readability\nplt.grid(True)\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:12.751218Z","iopub.execute_input":"2024-11-26T11:38:12.751577Z","iopub.status.idle":"2024-11-26T11:38:13.064394Z","shell.execute_reply.started":"2024-11-26T11:38:12.751541Z","shell.execute_reply":"2024-11-26T11:38:13.063319Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Scatter plot above shows that the Sleep disturbance scale of different people with SII 0, 1 and 2 is almost in similar range except a few outliers. \nWhile for SII 3 we have already seen above that we have very less entries so, this can explain why we have less mark/dots for it on the scatter plot. Even if we have less marks/dots for SII 3 they are still inside the same range in which we have the sleep diturbance values for SII 0, 1 and 2. Although the range of SII 3 is smaller as compared to SII 0, 1 and 2 but still I used the words same range because they don't have a big number of spots outside the range of SII 0, 1 and 2.\nA surprising point here is this that in case of SII 3 we don't have very high and very low sleep disturbance values. Ideally we would expect that SII 3 should have higher values for sleep disturbance but it is not the case here.","metadata":{}},{"cell_type":"code","source":"# Violin plot\nplt.figure(figsize=(8, 6))\n\n# Create the violin plot\nsns.violinplot(y='SDS-SDS_Total_Raw', data=train, color='royalblue')\n\n# Add labels and title\nplt.xlabel('SII')\nplt.ylabel('Sleep Disturbance Scale')\nplt.title('Violin Plot of Sleep Disturbance Scale Distribution')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T11:40:48.291644Z","iopub.execute_input":"2024-11-26T11:40:48.292029Z","iopub.status.idle":"2024-11-26T11:40:48.554798Z","shell.execute_reply.started":"2024-11-26T11:40:48.291995Z","shell.execute_reply":"2024-11-26T11:40:48.553688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Violin plot shows the distribution of data points. We see mostly data points are present between values 30 to 50 of sleep disturbance scale. It seems it is quite identical for sii values of 0, 1 and 2. While for sii value 3 it seems that data points are slightly less as compared to sii value 0, 1 and 2.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(18, 5))\n\n# SII by BIA_BMI\nsns.boxplot(y=train['BIA-BIA_BMI'], x=train['sii'], ax=axes[0], palette=\"Set2\")\naxes[0].set_title('SII by BIA BMI')\naxes[0].set_ylabel('BIA BMI')\naxes[0].set_xlabel('SII')\n\n# SII by Physical-BMI\nsns.boxplot(y=train['Physical-BMI'], x=train['sii'], ax=axes[1], palette=\"Set1\")\naxes[1].set_title('SII by Physical BMI')\naxes[1].set_ylabel('Physical BMI')\naxes[1].set_xlabel('SII')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.731562Z","iopub.status.idle":"2024-11-26T11:38:13.732218Z","shell.execute_reply.started":"2024-11-26T11:38:13.731847Z","shell.execute_reply":"2024-11-26T11:38:13.731882Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Recorded values for BIA BMI and Physical BMI seem slightly different. \n\n**There are two possible reasons for this:**\n1. There is some discrepancy or mistake in measurement of BMI.\n2. Both BMI's might have been calculated at different ages for same person.","metadata":{}},{"cell_type":"code","source":"DROP_COLS = [\n    \"Basic_Demos-Enroll_Season\", \"CGAS-Season\", \"Physical-Season\",\n    \"Fitness_Endurance-Season\", \"FGC-Season\", \"BIA-Season\",\n    \"PAQ_A-Season\", \"PAQ_C-Season\", \"SDS-Season\", \"PreInt_EduHx-Season\"\n]\n\ntrain = train.drop(DROP_COLS,axis=1)\ntest = test.drop(DROP_COLS,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.733933Z","iopub.status.idle":"2024-11-26T11:38:13.734497Z","shell.execute_reply.started":"2024-11-26T11:38:13.734209Z","shell.execute_reply":"2024-11-26T11:38:13.734239Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Season columns doesn't seem to contribute to our predictions so, we have dropped them.","metadata":{}},{"cell_type":"code","source":"for column in train.columns : \n    plt.figure(figsize = (14,4))\n    plt.subplot(121)\n    sns.histplot(train[column])\n    plt.title(column)\n    \n    plt.subplot(122)\n    stats.probplot(train[column],dist = 'norm', plot = plt)\n    plt.title(column)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.735988Z","iopub.status.idle":"2024-11-26T11:38:13.736562Z","shell.execute_reply.started":"2024-11-26T11:38:13.736279Z","shell.execute_reply":"2024-11-26T11:38:13.736309Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the graphs above it is clear that the columns for Bio-electric Impedance Analysis (BIA) are skewed so, lets apply Power Transformer with Yeo-Jhonson. \n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📈YEO-JHONSON</span>","metadata":{}},{"cell_type":"code","source":"# Identify columns to exclude\ninclude_cols = ['BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW']\n# Create copies of the train and test DataFrames to avoid modifying the originals\ntrain_transformed = train.copy()\ntest_transformed = test.copy()\ny_train = train['sii']\nX_train = train_transformed.drop('sii', axis=1)\n\n# Apply the Yeo-Johnson transformation to the numerical columns\npt = PowerTransformer()\n# Fit and transform the specified columns in the train DataFrame\nX_train[include_cols] = pt.fit_transform(X_train[include_cols])\n# Transform the same columns in the test DataFrame using the fitted transformer\ntest_transformed[include_cols] = pt.transform(test_transformed[include_cols])\n\ntest=test_transformed","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.740294Z","iopub.status.idle":"2024-11-26T11:38:13.741125Z","shell.execute_reply.started":"2024-11-26T11:38:13.740832Z","shell.execute_reply":"2024-11-26T11:38:13.740867Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in X_train[include_cols].columns : \n    plt.figure(figsize = (14,4))\n    plt.subplot(121)\n    sns.histplot(X_train[column])\n    plt.title(column)\n    \n    plt.subplot(122)\n    stats.probplot(X_train[column],dist = 'norm', plot = plt)\n    plt.title(column)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.742502Z","iopub.status.idle":"2024-11-26T11:38:13.743047Z","shell.execute_reply.started":"2024-11-26T11:38:13.742751Z","shell.execute_reply":"2024-11-26T11:38:13.742780Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"After applying Yeo-Jhonson, BIA columns have been normalised.\n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">✳️Sparce PCA</span>\nSparse Principal Component Analysis (Sparse PCA) is a dimensionality reduction technique that extends traditional PCA by introducing sparsity into the principal components. This means that each principal component is represented by only a few non-zero coefficients, making it easier to interpret and understand the underlying structure of the data.","metadata":{}},{"cell_type":"code","source":"# Standardize features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(test)\n\nn_components = 35  # Number of components to extract\nalpha = 0.01  # Sparsity parameter (higher alpha means sparser components)\n\n# Create a pipeline for imputation and Sparse PCA\npipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),  # Replace missing values with mean\n    ('sparse_pca', SparsePCA(n_components=n_components, alpha=alpha))  # Adjust n_components and alpha as needed\n])\n\n# Fit the pipeline to the training data\npipeline.fit(X_train_scaled)\n\n# Transform both training and test data\nX_train_sparse = pipeline.transform(X_train_scaled)\nX_test_sparse = pipeline.transform(X_test_scaled)\n\nX_train = pd.DataFrame(X_train_sparse)\nX_train = X_train.reset_index(drop=True)\ny_train = y_train.reset_index(drop=True)\n\n# Concatenate X_train and y_train\ntrain_data_pca = pd.concat([pd.DataFrame(X_train), y_train], axis=1)\n\nprint(train_data_pca.head(1))\n\ntrain = train_data_pca\ntest = pd.DataFrame(X_test_sparse)","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.744023Z","iopub.status.idle":"2024-11-26T11:38:13.744551Z","shell.execute_reply.started":"2024-11-26T11:38:13.744289Z","shell.execute_reply":"2024-11-26T11:38:13.744318Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.748025Z","iopub.status.idle":"2024-11-26T11:38:13.748575Z","shell.execute_reply.started":"2024-11-26T11:38:13.748305Z","shell.execute_reply":"2024-11-26T11:38:13.748334Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.749527Z","iopub.status.idle":"2024-11-26T11:38:13.750075Z","shell.execute_reply.started":"2024-11-26T11:38:13.749766Z","shell.execute_reply":"2024-11-26T11:38:13.749795Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">⚖️Quadratic Weighted Kappa</span>","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = model_class\n        \n        for name, model in model_class.items():\n            print(\"Name\")\n            print(name)\n            model.fit(X_train, y_train)\n\n        y_train_pred = blend_predict(X_train)\n        y_val_pred = blend_predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = blend_predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    print(KappaOPtimizer.x)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-11-26T11:38:13.752747Z","iopub.status.idle":"2024-11-26T11:38:13.753620Z","shell.execute_reply.started":"2024-11-26T11:38:13.753328Z","shell.execute_reply":"2024-11-26T11:38:13.753359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#edc8a8;color:black;padding:10px;border-radius:40px;\">📑Explaining Quadratic Weighted Kappa</span>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:white;color:black;padding:10px;border:20px groove brown;border-radius:40px;\">\n  <h2 style=\"color:black;\">Function <strong>quadratic_weighted_kappa</strong></h2>\n  <p style=\"color:black;\">This function calculates the quadratic weighted kappa score between two sets of labels (<strong>y_true</strong> and <strong>y_pred</strong>). It utilizes the <strong>cohen_kappa_score</strong> function from scikit-learn with the <strong>weights='quadratic'</strong> argument.</p>\n<strong>Code:</strong>\n<span>\n  def quadratic_weighted_kappa(y_true, y_pred):<br/>\n      &nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;return cohen_kappa_score(y_true, y_pred, weights='quadratic')</span>\n    \n<h2 style=\"color:black;\">Function <strong>TrainML</strong></h2>\n<p style=\"color:black;\">This function is responsible for training a machine learning model and evaluating its performance. It takes the model class (<strong>model_class</strong>) and testing data (<strong>test_data</strong>) as input.</p>\n\n<ul>\n<li>It separates the target variable (<strong>sii</strong>) from the features in the training data (<strong>X</strong> and <strong>y</strong>).</li>\n<li>It uses <strong>StratifiedKFold</strong> to perform stratified K-Fold cross-validation, ensuring balanced class distribution in each fold.</li>\n<li>It initializes arrays to store training and validation kappa scores (<strong>train_S</strong>, <strong>test_S</strong>), predicted probabilities (<strong>oof_non_rounded</strong>), rounded predictions (<strong>oof_rounded</strong>), and test set predictions (<strong>test_preds</strong>).</li>\n<li>It iterates through each fold using a progress bar (<strong>tqdm</strong>).</li>\n<ul>\n<li>Within each fold, it splits the training data into training and validation sets (<strong>X_train</strong>, <strong>X_val</strong>, <strong>y_train</strong>, <strong>y_val</strong>).</li>\n<li>It iterates through each model within the provided model class (<strong>model_class</strong>).</li>\n<ul>\n<li>It prints the model name for debugging purposes.</li>\n<li>It fits the model on the training data (<strong>model.fit(X_train, y_train)</strong>).</li>\n</ul>\n<li>It predicts probabilities for both the training and validation sets using a blending function (<strong>blend_predict</strong>).</li>\n<li>It stores the predicted probabilities for the validation set (<strong>oof_non_rounded</strong>) and rounds them to integers for evaluation (<strong>oof_rounded</strong>).</li>\n<li>It calculates and stores the quadratic weighted kappa scores for the training and validation sets (<strong>train_kappa</strong>, <strong>val_kappa</strong>).</li>\n<li>It appends these kappa scores to their respective lists (<strong>train_S</strong>, <strong>test_S</strong>).</li>\n<li>It predicts probabilities for the test data using the blending function and stores them (<strong>test_preds[:, fold] = blend_predict(test_data)</strong>).</li>\n<li>It prints the fold number, training kappa score, and validation kappa score.</li>\n</ul>\n<li>It prints the mean training and validation kappa scores.</li>\n<li>It defines an optimization problem using <strong>minimize</strong> to find the optimal thresholds for rounding the predicted probabilities. This optimization minimizes the negative quadratic weighted kappa (<strong>evaluate_predictions</strong>) on the validation set (<strong>y</strong>, <strong>oof_non_rounded</strong>). The Nelder-Mead method is used for optimization.</li>\n<li>It asserts that the optimization converged successfully.</li>\n<li>It applies the optimized thresholds to the validation set probabilities (<strong>oof_tuned</strong>) and calculates the final kappa score (<strong>tKappa</strong>).</li>\n<li>It prints the optimized kappa score.</li>\n<li>It calculates the mean of the predicted probabilities across all folds for the test set (<strong>tpm</strong>).</li>\n<li>It retrieves the optimized thresholds from the optimization result (<strong>KappaOPtimizer.x</strong>).</li>\n<li>It applies the optimized thresholds to the mean test set probabilities (<strong>tpTuned</strong>).</li>\n<li>Finally, it creates a submission DataFrame with the IDs from the sample data and the rounded predictions (<strong>tpTuned</strong>) and returns this DataFrame along with the trained model.</li>\n</ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🖥️Pipelines and Models</span>","metadata":{}},{"cell_type":"code","source":"# Create pipelines for  Catboost Regression, LightGBM Regression, and XGBoost Regression\nSEED = 5\n\n#Using LGBMRegressor\nParams_lgb = {\n    'learning_rate': 0.01, \n    'max_depth': 12, \n    'num_leaves': 413, \n    'min_data_in_leaf': 14,\n    'feature_fraction': 0.7987976913702801, \n    'bagging_fraction': 0.7602261703576205, \n    'bagging_freq': 2, \n    'lambda_l1': 5.5, \n    'lambda_l2': 5.5e-06\n} \n\nLight = lgb.LGBMRegressor(**Params_lgb,random_state=SEED, verbose=-1,n_estimators=585)\n\n\n# Using XGBRegressor\nParams_xgb = {\n    'learning_rate': 0.01,\n    'max_depth': 12,\n    'max_leaves': 413,\n    'min_child_weight': 14,\n    'colsample_bytree': 0.7987976913702801,\n    'subsample': 0.7602261703576205,\n    'reg_alpha': 5.5,\n    'reg_lambda': 5.5e-06\n}\n\nXGBoost = xgb.XGBRegressor(**Params_xgb, random_state=SEED, verbosity=0, n_estimators=585)\n\n\n# Using CatBoostRegressor\nParams_cat = {\n    'learning_rate': 0.01,\n    'max_depth': 12,\n    'min_child_samples': 14,  # Equivalent to min_data_in_leaf\n    'rsm': 0.7987976913702801,  # Equivalent to feature_fraction\n    'subsample': 0.7602261703576205,  # Equivalent to bagging_fraction\n    'l2_leaf_reg': 5.5,  # Equivalent to lambda_l2/lambda_l1\n    'random_seed': SEED,  # Use same seed for reproducibility\n    'n_estimators': 585,  # Number of trees\n    'verbose': 0  # Suppresses output\n}\n\nCatBoost = CatBoostRegressor(**Params_cat)","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.754630Z","iopub.status.idle":"2024-11-26T11:38:13.755156Z","shell.execute_reply.started":"2024-11-26T11:38:13.754894Z","shell.execute_reply":"2024-11-26T11:38:13.754922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📕Create Dictionaries and Predict</span>","metadata":{}},{"cell_type":"code","source":"# Create a dictionary to store the models\nmodels = {'XGBoostRegressor': XGBoost, \n          'LightGBMRegressor': Light, \n          'CatBoostRegressor': CatBoost}\n\ndef blend_predict(X):\n    return ((0.45 * Light.predict(X)) +\n            (0.3 * XGBoost.predict(X)) +\n            (0.25 * CatBoost.predict(X)))","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.756401Z","iopub.status.idle":"2024-11-26T11:38:13.757013Z","shell.execute_reply.started":"2024-11-26T11:38:13.756711Z","shell.execute_reply":"2024-11-26T11:38:13.756741Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🚅Train Model</span>","metadata":{}},{"cell_type":"code","source":"Submission, model = TrainML(models, test)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-11-26T11:38:13.758355Z","iopub.status.idle":"2024-11-26T11:38:13.758887Z","shell.execute_reply.started":"2024-11-26T11:38:13.758611Z","shell.execute_reply":"2024-11-26T11:38:13.758638Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📁Submission</span>","metadata":{}},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.760198Z","iopub.status.idle":"2024-11-26T11:38:13.760798Z","shell.execute_reply.started":"2024-11-26T11:38:13.760447Z","shell.execute_reply":"2024-11-26T11:38:13.760551Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-11-26T11:38:13.762606Z","iopub.status.idle":"2024-11-26T11:38:13.763137Z","shell.execute_reply.started":"2024-11-26T11:38:13.762864Z","shell.execute_reply":"2024-11-26T11:38:13.762947Z"},"trusted":true},"outputs":[],"execution_count":null}]}