{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#1. Loading the Datasets\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom scipy.stats import zscore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:54:19.061891Z","iopub.execute_input":"2024-12-06T03:54:19.062343Z","iopub.status.idle":"2024-12-06T03:54:19.068989Z","shell.execute_reply.started":"2024-12-06T03:54:19.062308Z","shell.execute_reply":"2024-12-06T03:54:19.067802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = '/kaggle/input/child-mind-institute-problematic-internet-use'\n\ntrain_df = pd.read_csv(f'{root_dir}/train.csv')\ntest_df = pd.read_csv(f'{root_dir}/test.csv')\nsample_submission_df = pd.read_csv(f'{root_dir}/sample_submission.csv')\ndata_dict_df = pd.read_csv(f'{root_dir}/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:54:21.926948Z","iopub.execute_input":"2024-12-06T03:54:21.927389Z","iopub.status.idle":"2024-12-06T03:54:22.006513Z","shell.execute_reply.started":"2024-12-06T03:54:21.927354Z","shell.execute_reply":"2024-12-06T03:54:22.005344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info(), test_df.info(), sample_submission_df.info(), data_dict_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:54:25.679622Z","iopub.execute_input":"2024-12-06T03:54:25.680824Z","iopub.status.idle":"2024-12-06T03:54:25.718782Z","shell.execute_reply.started":"2024-12-06T03:54:25.680778Z","shell.execute_reply":"2024-12-06T03:54:25.717665Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**2. Basic Cleaning**","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:35.782447Z","iopub.execute_input":"2024-12-06T03:18:35.782907Z","iopub.status.idle":"2024-12-06T03:18:35.788440Z","shell.execute_reply.started":"2024-12-06T03:18:35.782873Z","shell.execute_reply":"2024-12-06T03:18:35.787082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']\ncategorical_features = ['Basic_Demos-Sex', 'CGAS-Season', 'Physical-Season', 'Basic_Demos-Enroll_Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:37.074824Z","iopub.execute_input":"2024-12-06T03:18:37.075215Z","iopub.status.idle":"2024-12-06T03:18:37.081203Z","shell.execute_reply.started":"2024-12-06T03:18:37.075182Z","shell.execute_reply":"2024-12-06T03:18:37.079910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_imputer = SimpleImputer(strategy='mean')\ncategorical_imputer = SimpleImputer(strategy='most_frequent')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:38.408642Z","iopub.execute_input":"2024-12-06T03:18:38.409852Z","iopub.status.idle":"2024-12-06T03:18:38.414795Z","shell.execute_reply.started":"2024-12-06T03:18:38.409802Z","shell.execute_reply":"2024-12-06T03:18:38.413604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[numeric_features] = numeric_imputer.fit_transform(train_df[numeric_features])\ntrain_df[categorical_features] = categorical_imputer.fit_transform(train_df[categorical_features])\n\ntest_df[numeric_features] = numeric_imputer.transform(test_df[numeric_features])\ntest_df[categorical_features] = categorical_imputer.transform(test_df[categorical_features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:39.629120Z","iopub.execute_input":"2024-12-06T03:18:39.629543Z","iopub.status.idle":"2024-12-06T03:18:39.662372Z","shell.execute_reply.started":"2024-12-06T03:18:39.629508Z","shell.execute_reply":"2024-12-06T03:18:39.661062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum(), test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:41.121287Z","iopub.execute_input":"2024-12-06T03:18:41.121677Z","iopub.status.idle":"2024-12-06T03:18:41.137802Z","shell.execute_reply.started":"2024-12-06T03:18:41.121645Z","shell.execute_reply":"2024-12-06T03:18:41.136541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**3. Feature Engineering**","metadata":{}},{"cell_type":"code","source":"train_df['BMI_Category'] = pd.cut(train_df['Physical-BMI'], bins=[0, 18.5, 24.9, 29.9, np.inf], labels=['Underweight', 'Normal', 'Overweight', 'Obese'])\ntest_df['BMI_Category'] = pd.cut(test_df['Physical-BMI'], bins=[0, 18.5, 24.9, 29.9, np.inf], labels=['Underweight', 'Normal', 'Overweight', 'Obese'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:44.674339Z","iopub.execute_input":"2024-12-06T03:18:44.674824Z","iopub.status.idle":"2024-12-06T03:18:44.691017Z","shell.execute_reply.started":"2024-12-06T03:18:44.674788Z","shell.execute_reply":"2024-12-06T03:18:44.689449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Age_Group'] = pd.cut(train_df['Basic_Demos-Age'], bins=[0, 18, 30, 40, np.inf], labels=['Child', 'Young Adult', 'Adult', 'Senior'])\ntest_df['Age_Group'] = pd.cut(test_df['Basic_Demos-Age'], bins=[0, 18, 30, 40, np.inf], labels=['Child', 'Young Adult', 'Adult', 'Senior'])\n\n\ntrain_df[['Physical-BMI', 'BMI_Category', 'Basic_Demos-Age', 'Age_Group']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:45.896862Z","iopub.execute_input":"2024-12-06T03:18:45.897342Z","iopub.status.idle":"2024-12-06T03:18:45.921297Z","shell.execute_reply.started":"2024-12-06T03:18:45.897305Z","shell.execute_reply":"2024-12-06T03:18:45.920080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['BMI_Category'] = pd.cut(\n    train_df['Physical-BMI'], \n    bins=[0, 18.5, 24.9, 29.9, np.inf], \n    labels=['Underweight', 'Normal', 'Overweight', 'Obese']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:47.794066Z","iopub.execute_input":"2024-12-06T03:18:47.794497Z","iopub.status.idle":"2024-12-06T03:18:47.804033Z","shell.execute_reply.started":"2024-12-06T03:18:47.794462Z","shell.execute_reply":"2024-12-06T03:18:47.802831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['BMI_Category'] = pd.cut(\n    test_df['Physical-BMI'], \n    bins=[0, 18.5, 24.9, 29.9, np.inf], \n    labels=['Underweight', 'Normal', 'Overweight', 'Obese']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:49.050103Z","iopub.execute_input":"2024-12-06T03:18:49.050535Z","iopub.status.idle":"2024-12-06T03:18:49.058382Z","shell.execute_reply.started":"2024-12-06T03:18:49.050503Z","shell.execute_reply":"2024-12-06T03:18:49.056832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Age_Group'] = pd.cut(\n    train_df['Basic_Demos-Age'], \n    bins=[0, 18, 30, 40, np.inf], \n    labels=['Child', 'Young Adult', 'Adult', 'Senior']\n)\ntest_df['Age_Group'] = pd.cut(\n    test_df['Basic_Demos-Age'], \n    bins=[0, 18, 30, 40, np.inf], \n    labels=['Child', 'Young Adult', 'Adult', 'Senior']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:50.352033Z","iopub.execute_input":"2024-12-06T03:18:50.353266Z","iopub.status.idle":"2024-12-06T03:18:50.363752Z","shell.execute_reply.started":"2024-12-06T03:18:50.353225Z","shell.execute_reply":"2024-12-06T03:18:50.362139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**4. Scaling and Encoding**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:52.584818Z","iopub.execute_input":"2024-12-06T03:18:52.585981Z","iopub.status.idle":"2024-12-06T03:18:52.591040Z","shell.execute_reply.started":"2024-12-06T03:18:52.585942Z","shell.execute_reply":"2024-12-06T03:18:52.589661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:53.761740Z","iopub.execute_input":"2024-12-06T03:18:53.762149Z","iopub.status.idle":"2024-12-06T03:18:53.768384Z","shell.execute_reply.started":"2024-12-06T03:18:53.762114Z","shell.execute_reply":"2024-12-06T03:18:53.766883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:54.885137Z","iopub.execute_input":"2024-12-06T03:18:54.885769Z","iopub.status.idle":"2024-12-06T03:18:54.892248Z","shell.execute_reply.started":"2024-12-06T03:18:54.885698Z","shell.execute_reply":"2024-12-06T03:18:54.890674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:56.085351Z","iopub.execute_input":"2024-12-06T03:18:56.085790Z","iopub.status.idle":"2024-12-06T03:18:56.091543Z","shell.execute_reply.started":"2024-12-06T03:18:56.085740Z","shell.execute_reply":"2024-12-06T03:18:56.090299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']\ncategorical_features = ['Basic_Demos-Sex', 'CGAS-Season', 'Physical-Season', 'Basic_Demos-Enroll_Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:57.621629Z","iopub.execute_input":"2024-12-06T03:18:57.622110Z","iopub.status.idle":"2024-12-06T03:18:57.628358Z","shell.execute_reply.started":"2024-12-06T03:18:57.622069Z","shell.execute_reply":"2024-12-06T03:18:57.626786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\nencoder = OneHotEncoder(drop='first', sparse_output=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:18:59.136789Z","iopub.execute_input":"2024-12-06T03:18:59.137282Z","iopub.status.idle":"2024-12-06T03:18:59.143573Z","shell.execute_reply.started":"2024-12-06T03:18:59.137243Z","shell.execute_reply":"2024-12-06T03:18:59.141982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', scaler, numeric_features),    \n        ('cat', encoder, categorical_features)  \n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:00.249839Z","iopub.execute_input":"2024-12-06T03:19:00.250258Z","iopub.status.idle":"2024-12-06T03:19:00.256520Z","shell.execute_reply.started":"2024-12-06T03:19:00.250223Z","shell.execute_reply":"2024-12-06T03:19:00.255245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = OneHotEncoder(drop='first', sparse_output=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:01.535508Z","iopub.execute_input":"2024-12-06T03:19:01.535983Z","iopub.status.idle":"2024-12-06T03:19:01.541945Z","shell.execute_reply.started":"2024-12-06T03:19:01.535947Z","shell.execute_reply":"2024-12-06T03:19:01.540692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_processed = preprocessor.fit_transform(train_df)\nX_test_processed = preprocessor.transform(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:02.995423Z","iopub.execute_input":"2024-12-06T03:19:02.995946Z","iopub.status.idle":"2024-12-06T03:19:03.029219Z","shell.execute_reply.started":"2024-12-06T03:19:02.995906Z","shell.execute_reply":"2024-12-06T03:19:03.027980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder_fitted = preprocessor.named_transformers_['cat']  \nprocessed_columns = numeric_features + list(encoder_fitted.get_feature_names_out(categorical_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:04.418283Z","iopub.execute_input":"2024-12-06T03:19:04.418816Z","iopub.status.idle":"2024-12-06T03:19:04.426064Z","shell.execute_reply.started":"2024-12-06T03:19:04.418774Z","shell.execute_reply":"2024-12-06T03:19:04.424481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']\ncategorical_features = ['Basic_Demos-Sex', 'CGAS-Season', 'Physical-Season', 'Basic_Demos-Enroll_Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:05.977415Z","iopub.execute_input":"2024-12-06T03:19:05.977967Z","iopub.status.idle":"2024-12-06T03:19:05.983964Z","shell.execute_reply.started":"2024-12-06T03:19:05.977926Z","shell.execute_reply":"2024-12-06T03:19:05.982656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_df = pd.DataFrame(X_train_processed, columns=processed_columns)\nX_test_df = pd.DataFrame(X_test_processed, columns=processed_columns)\n\nX_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:07.223157Z","iopub.execute_input":"2024-12-06T03:19:07.224466Z","iopub.status.idle":"2024-12-06T03:19:07.249086Z","shell.execute_reply.started":"2024-12-06T03:19:07.224420Z","shell.execute_reply":"2024-12-06T03:19:07.247614Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**5. Outlier Detection**","metadata":{}},{"cell_type":"code","source":"def detect_outliers(df, features):\n    outlier_indices = []\n    for feature in features:\n        Q1 = df[feature].quantile(0.25)\n        Q3 = df[feature].quantile(0.75)\n        IQR = Q3 - Q1\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n        outliers = df[(df[feature] < lower_bound) | (df[feature] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n    return set(outlier_indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:09.145002Z","iopub.execute_input":"2024-12-06T03:19:09.145395Z","iopub.status.idle":"2024-12-06T03:19:09.152951Z","shell.execute_reply.started":"2024-12-06T03:19:09.145362Z","shell.execute_reply":"2024-12-06T03:19:09.151735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outliers = detect_outliers(train_df, numeric_features)\ntrain_df_cleaned = train_df.drop(index=outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:10.962398Z","iopub.execute_input":"2024-12-06T03:19:10.962811Z","iopub.status.idle":"2024-12-06T03:19:10.992515Z","shell.execute_reply.started":"2024-12-06T03:19:10.962778Z","shell.execute_reply":"2024-12-06T03:19:10.991140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Outliers detected: {len(outliers)}')\ntrain_df_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:12.028437Z","iopub.execute_input":"2024-12-06T03:19:12.028945Z","iopub.status.idle":"2024-12-06T03:19:12.054213Z","shell.execute_reply.started":"2024-12-06T03:19:12.028899Z","shell.execute_reply":"2024-12-06T03:19:12.052709Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**6. Data Visualization**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:15.403957Z","iopub.execute_input":"2024-12-06T03:19:15.405183Z","iopub.status.idle":"2024-12-06T03:19:15.409890Z","shell.execute_reply.started":"2024-12-06T03:19:15.405140Z","shell.execute_reply":"2024-12-06T03:19:15.408799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']].hist(bins=20, figsize=(10, 6))\nplt.suptitle('Distribution of Numeric Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:16.377480Z","iopub.execute_input":"2024-12-06T03:19:16.377970Z","iopub.status.idle":"2024-12-06T03:19:17.352372Z","shell.execute_reply.started":"2024-12-06T03:19:16.377933Z","shell.execute_reply":"2024-12-06T03:19:17.351044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = train_df[numeric_features].corr()\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title('Correlation Heatmap of Numeric Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T02:46:42.159000Z","iopub.execute_input":"2024-12-05T02:46:42.159489Z","iopub.status.idle":"2024-12-05T02:46:42.904220Z","shell.execute_reply.started":"2024-12-05T02:46:42.159452Z","shell.execute_reply":"2024-12-05T02:46:42.902967Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**7. Training our model**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:26.003415Z","iopub.execute_input":"2024-12-06T03:19:26.003828Z","iopub.status.idle":"2024-12-06T03:19:26.116703Z","shell.execute_reply.started":"2024-12-06T03:19:26.003795Z","shell.execute_reply":"2024-12-06T03:19:26.115099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = train_df['Physical-BMI']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:21.382988Z","iopub.execute_input":"2024-12-06T03:19:21.383373Z","iopub.status.idle":"2024-12-06T03:19:21.388938Z","shell.execute_reply.started":"2024-12-06T03:19:21.383342Z","shell.execute_reply":"2024-12-06T03:19:21.387781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:22.626573Z","iopub.execute_input":"2024-12-06T03:19:22.627042Z","iopub.status.idle":"2024-12-06T03:19:22.632606Z","shell.execute_reply.started":"2024-12-06T03:19:22.627007Z","shell.execute_reply":"2024-12-06T03:19:22.631413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(\n    X_train_df, y_train, test_size=0.2, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:23.910714Z","iopub.execute_input":"2024-12-06T03:19:23.911146Z","iopub.status.idle":"2024-12-06T03:19:23.921423Z","shell.execute_reply.started":"2024-12-06T03:19:23.911112Z","shell.execute_reply":"2024-12-06T03:19:23.920139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train_df.shape)  \nprint(y_train.shape) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:27.922199Z","iopub.execute_input":"2024-12-06T03:19:27.922585Z","iopub.status.idle":"2024-12-06T03:19:27.929745Z","shell.execute_reply.started":"2024-12-06T03:19:27.922553Z","shell.execute_reply":"2024-12-06T03:19:27.928069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:29.711491Z","iopub.execute_input":"2024-12-06T03:19:29.712280Z","iopub.status.idle":"2024-12-06T03:19:29.717517Z","shell.execute_reply.started":"2024-12-06T03:19:29.712241Z","shell.execute_reply":"2024-12-06T03:19:29.716055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:31.185037Z","iopub.execute_input":"2024-12-06T03:19:31.185438Z","iopub.status.idle":"2024-12-06T03:19:31.191210Z","shell.execute_reply.started":"2024-12-06T03:19:31.185406Z","shell.execute_reply":"2024-12-06T03:19:31.189661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrf_regressor = RandomForestRegressor(random_state=42)\nrf_regressor.fit(X_train, y_train)\ny_pred = rf_regressor.predict(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:32.619138Z","iopub.execute_input":"2024-12-06T03:19:32.619603Z","iopub.status.idle":"2024-12-06T03:19:33.900980Z","shell.execute_reply.started":"2024-12-06T03:19:32.619565Z","shell.execute_reply":"2024-12-06T03:19:33.899793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:35.441473Z","iopub.execute_input":"2024-12-06T03:19:35.441898Z","iopub.status.idle":"2024-12-06T03:19:35.501062Z","shell.execute_reply.started":"2024-12-06T03:19:35.441863Z","shell.execute_reply":"2024-12-06T03:19:35.499692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:36.968676Z","iopub.execute_input":"2024-12-06T03:19:36.969099Z","iopub.status.idle":"2024-12-06T03:19:36.975200Z","shell.execute_reply.started":"2024-12-06T03:19:36.969065Z","shell.execute_reply":"2024-12-06T03:19:36.973806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mae = mean_absolute_error(y_valid, y_pred)\nmse = mean_squared_error(y_valid, y_pred)\nrmse = mean_squared_error(y_valid, y_pred, squared=False)  \nr2 = r2_score(y_valid, y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:38.346801Z","iopub.execute_input":"2024-12-06T03:19:38.347254Z","iopub.status.idle":"2024-12-06T03:19:38.357940Z","shell.execute_reply.started":"2024-12-06T03:19:38.347218Z","shell.execute_reply":"2024-12-06T03:19:38.356653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"R-squared (R²): {r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:39.607965Z","iopub.execute_input":"2024-12-06T03:19:39.608412Z","iopub.status.idle":"2024-12-06T03:19:39.615381Z","shell.execute_reply.started":"2024-12-06T03:19:39.608377Z","shell.execute_reply":"2024-12-06T03:19:39.614234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"residuals = y_valid - y_pred\n\nplt.figure(figsize=(8, 6))\nplt.scatter(y_pred, residuals, alpha=0.6)\nplt.axhline(0, color='red', linestyle='--')\nplt.xlabel(\"Predicted Values\")\nplt.ylabel(\"Residuals\")\nplt.title(\"Residual Plot\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:41.270968Z","iopub.execute_input":"2024-12-06T03:19:41.271440Z","iopub.status.idle":"2024-12-06T03:19:41.712450Z","shell.execute_reply.started":"2024-12-06T03:19:41.271404Z","shell.execute_reply":"2024-12-06T03:19:41.711128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:43.845934Z","iopub.execute_input":"2024-12-06T03:19:43.846386Z","iopub.status.idle":"2024-12-06T03:19:43.852172Z","shell.execute_reply.started":"2024-12-06T03:19:43.846350Z","shell.execute_reply":"2024-12-06T03:19:43.850896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:45.141557Z","iopub.execute_input":"2024-12-06T03:19:45.142001Z","iopub.status.idle":"2024-12-06T03:19:45.148009Z","shell.execute_reply.started":"2024-12-06T03:19:45.141965Z","shell.execute_reply":"2024-12-06T03:19:45.146416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:46.562019Z","iopub.execute_input":"2024-12-06T03:19:46.562454Z","iopub.status.idle":"2024-12-06T03:19:46.569198Z","shell.execute_reply.started":"2024-12-06T03:19:46.562418Z","shell.execute_reply":"2024-12-06T03:19:46.567548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ngrid_search = GridSearchCV(estimator=RandomForestRegressor(random_state=42),\n                           param_grid=param_grid,\n                           cv=3, n_jobs=-1, verbose=2, scoring='neg_mean_squared_error')\ngrid_search.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:19:48.104850Z","iopub.execute_input":"2024-12-06T03:19:48.105297Z","iopub.status.idle":"2024-12-06T03:22:42.130437Z","shell.execute_reply.started":"2024-12-06T03:19:48.105262Z","shell.execute_reply":"2024-12-06T03:22:42.129056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = grid_search.best_estimator_\nprint(f\"Best Parameters: {grid_search.best_params_}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:20.958254Z","iopub.execute_input":"2024-12-06T03:23:20.959379Z","iopub.status.idle":"2024-12-06T03:23:20.965837Z","shell.execute_reply.started":"2024-12-06T03:23:20.959328Z","shell.execute_reply":"2024-12-06T03:23:20.964587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:23.381368Z","iopub.execute_input":"2024-12-06T03:23:23.382508Z","iopub.status.idle":"2024-12-06T03:23:23.388626Z","shell.execute_reply.started":"2024-12-06T03:23:23.382452Z","shell.execute_reply":"2024-12-06T03:23:23.387461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:24.800063Z","iopub.execute_input":"2024-12-06T03:23:24.800444Z","iopub.status.idle":"2024-12-06T03:23:24.824522Z","shell.execute_reply.started":"2024-12-06T03:23:24.800413Z","shell.execute_reply":"2024-12-06T03:23:24.823056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualizations","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:27.977245Z","iopub.execute_input":"2024-12-06T03:23:27.978260Z","iopub.status.idle":"2024-12-06T03:23:27.982839Z","shell.execute_reply.started":"2024-12-06T03:23:27.978219Z","shell.execute_reply":"2024-12-06T03:23:27.981775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train_df['Physical-BMI'], bins=20, kde=True, color=\"skyblue\", alpha=0.7)\nplt.title(\"Histogram of Target Variable: Physical-BMI\")\nplt.xlabel(\"Physical-BMI\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:29.449266Z","iopub.execute_input":"2024-12-06T03:23:29.450845Z","iopub.status.idle":"2024-12-06T03:23:29.909151Z","shell.execute_reply.started":"2024-12-06T03:23:29.450774Z","shell.execute_reply":"2024-12-06T03:23:29.907795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nsns.countplot(x=train_df['BMI_Category'], palette=\"pastel\")\nplt.title(\"Class Distribution of BMI Categories\")\nplt.xlabel(\"BMI Category\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:32.153323Z","iopub.execute_input":"2024-12-06T03:23:32.153805Z","iopub.status.idle":"2024-12-06T03:23:32.424410Z","shell.execute_reply.started":"2024-12-06T03:23:32.153764Z","shell.execute_reply":"2024-12-06T03:23:32.423247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"category_proportions = train_df['BMI_Category'].value_counts(normalize=True) * 100\nprint(\"Proportions of Each BMI Category (%):\\n\", category_proportions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:34.498626Z","iopub.execute_input":"2024-12-06T03:23:34.499395Z","iopub.status.idle":"2024-12-06T03:23:34.508579Z","shell.execute_reply.started":"2024-12-06T03:23:34.499356Z","shell.execute_reply":"2024-12-06T03:23:34.507149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model evaluation ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, roc_auc_score, roc_curve, precision_score, recall_score, f1_score\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:38.189929Z","iopub.execute_input":"2024-12-06T03:23:38.191021Z","iopub.status.idle":"2024-12-06T03:23:38.196916Z","shell.execute_reply.started":"2024-12-06T03:23:38.190975Z","shell.execute_reply":"2024-12-06T03:23:38.195318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_valid_class = pd.cut(\n    y_valid, \n    bins=[0, 18.5, 24.9, 29.9, np.inf], \n    labels=['Underweight', 'Normal', 'Overweight', 'Obese']\n)\n\ny_pred_class = pd.cut(\n    y_pred, \n    bins=[0, 18.5, 24.9, 29.9, np.inf], \n    labels=['Underweight', 'Normal', 'Overweight', 'Obese']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:39.650370Z","iopub.execute_input":"2024-12-06T03:23:39.650873Z","iopub.status.idle":"2024-12-06T03:23:39.661085Z","shell.execute_reply.started":"2024-12-06T03:23:39.650833Z","shell.execute_reply":"2024-12-06T03:23:39.659855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Classification Report:\")\nprint(classification_report(y_valid_class, y_pred_class))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:41.347794Z","iopub.execute_input":"2024-12-06T03:23:41.348267Z","iopub.status.idle":"2024-12-06T03:23:41.386980Z","shell.execute_reply.started":"2024-12-06T03:23:41.348230Z","shell.execute_reply":"2024-12-06T03:23:41.385569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"precision = precision_score(y_valid_class, y_pred_class, average='weighted')\nrecall = recall_score(y_valid_class, y_pred_class, average='weighted')\nf1 = f1_score(y_valid_class, y_pred_class, average='weighted')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:50.050627Z","iopub.execute_input":"2024-12-06T03:23:50.051234Z","iopub.status.idle":"2024-12-06T03:23:50.094332Z","shell.execute_reply.started":"2024-12-06T03:23:50.051193Z","shell.execute_reply":"2024-12-06T03:23:50.092623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Precision (Weighted): {precision:.4f}\")\nprint(f\"Recall (Weighted): {recall:.4f}\")\nprint(f\"F1-Score (Weighted): {f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:23:56.577143Z","iopub.execute_input":"2024-12-06T03:23:56.577596Z","iopub.status.idle":"2024-12-06T03:23:56.585421Z","shell.execute_reply.started":"2024-12-06T03:23:56.577558Z","shell.execute_reply":"2024-12-06T03:23:56.584000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:24:00.107330Z","iopub.execute_input":"2024-12-06T03:24:00.107819Z","iopub.status.idle":"2024-12-06T03:24:00.113577Z","shell.execute_reply.started":"2024-12-06T03:24:00.107782Z","shell.execute_reply":"2024-12-06T03:24:00.112135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mae = mean_absolute_error(y_valid, y_pred)\nmse = mean_squared_error(y_valid, y_pred)\nrmse = mean_squared_error(y_valid, y_pred, squared=False)  \nr2 = r2_score(y_valid, y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:24:12.670752Z","iopub.execute_input":"2024-12-06T03:24:12.671161Z","iopub.status.idle":"2024-12-06T03:24:12.682056Z","shell.execute_reply.started":"2024-12-06T03:24:12.671127Z","shell.execute_reply":"2024-12-06T03:24:12.680630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"R-squared (R²): {r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:24:16.919293Z","iopub.execute_input":"2024-12-06T03:24:16.919815Z","iopub.status.idle":"2024-12-06T03:24:16.927555Z","shell.execute_reply.started":"2024-12-06T03:24:16.919777Z","shell.execute_reply":"2024-12-06T03:24:16.926010Z"}},"outputs":[],"execution_count":null}]}