{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#1. Loading the Datasets\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom scipy.stats import zscore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:17.527239Z","iopub.execute_input":"2024-12-04T07:49:17.527652Z","iopub.status.idle":"2024-12-04T07:49:21.438841Z","shell.execute_reply.started":"2024-12-04T07:49:17.527604Z","shell.execute_reply":"2024-12-04T07:49:21.437651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = '/kaggle/input/child-mind-institute-problematic-internet-use'\n\ntrain_df = pd.read_csv(f'{root_dir}/train.csv')\ntest_df = pd.read_csv(f'{root_dir}/test.csv')\nsample_submission_df = pd.read_csv(f'{root_dir}/sample_submission.csv')\ndata_dict_df = pd.read_csv(f'{root_dir}/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:24.967299Z","iopub.execute_input":"2024-12-04T07:49:24.967868Z","iopub.status.idle":"2024-12-04T07:49:25.084778Z","shell.execute_reply.started":"2024-12-04T07:49:24.967830Z","shell.execute_reply":"2024-12-04T07:49:25.083646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info(), test_df.info(), sample_submission_df.info(), data_dict_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:28.052642Z","iopub.execute_input":"2024-12-04T07:49:28.053025Z","iopub.status.idle":"2024-12-04T07:49:28.115636Z","shell.execute_reply.started":"2024-12-04T07:49:28.052994Z","shell.execute_reply":"2024-12-04T07:49:28.114389Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**2. Basic Cleaning**","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:33.959168Z","iopub.execute_input":"2024-12-04T07:49:33.959555Z","iopub.status.idle":"2024-12-04T07:49:33.965037Z","shell.execute_reply.started":"2024-12-04T07:49:33.959507Z","shell.execute_reply":"2024-12-04T07:49:33.963831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_features = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']\ncategorical_features = ['Basic_Demos-Sex', 'CGAS-Season', 'Physical-Season', 'Basic_Demos-Enroll_Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:35.877582Z","iopub.execute_input":"2024-12-04T07:49:35.878006Z","iopub.status.idle":"2024-12-04T07:49:35.883542Z","shell.execute_reply.started":"2024-12-04T07:49:35.877970Z","shell.execute_reply":"2024-12-04T07:49:35.882262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_imputer = SimpleImputer(strategy='mean')\ncategorical_imputer = SimpleImputer(strategy='most_frequent')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:37.374630Z","iopub.execute_input":"2024-12-04T07:49:37.375046Z","iopub.status.idle":"2024-12-04T07:49:37.380696Z","shell.execute_reply.started":"2024-12-04T07:49:37.375011Z","shell.execute_reply":"2024-12-04T07:49:37.379317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[numeric_features] = numeric_imputer.fit_transform(train_df[numeric_features])\ntrain_df[categorical_features] = categorical_imputer.fit_transform(train_df[categorical_features])\n\ntest_df[numeric_features] = numeric_imputer.transform(test_df[numeric_features])\ntest_df[categorical_features] = categorical_imputer.transform(test_df[categorical_features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:38.826355Z","iopub.execute_input":"2024-12-04T07:49:38.826762Z","iopub.status.idle":"2024-12-04T07:49:38.858401Z","shell.execute_reply.started":"2024-12-04T07:49:38.826725Z","shell.execute_reply":"2024-12-04T07:49:38.857343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum(), test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:40.625927Z","iopub.execute_input":"2024-12-04T07:49:40.626339Z","iopub.status.idle":"2024-12-04T07:49:40.642744Z","shell.execute_reply.started":"2024-12-04T07:49:40.626304Z","shell.execute_reply":"2024-12-04T07:49:40.641652Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**3. Feature Engineering**","metadata":{}},{"cell_type":"code","source":"train_df['BMI_Category'] = pd.cut(train_df['Physical-BMI'], bins=[0, 18.5, 24.9, 29.9, np.inf], labels=['Underweight', 'Normal', 'Overweight', 'Obese'])\ntest_df['BMI_Category'] = pd.cut(test_df['Physical-BMI'], bins=[0, 18.5, 24.9, 29.9, np.inf], labels=['Underweight', 'Normal', 'Overweight', 'Obese'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:45.109651Z","iopub.execute_input":"2024-12-04T07:49:45.110038Z","iopub.status.idle":"2024-12-04T07:49:45.125544Z","shell.execute_reply.started":"2024-12-04T07:49:45.110000Z","shell.execute_reply":"2024-12-04T07:49:45.124200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['Age_Group'] = pd.cut(train_df['Basic_Demos-Age'], bins=[0, 18, 30, 40, np.inf], labels=['Child', 'Young Adult', 'Adult', 'Senior'])\ntest_df['Age_Group'] = pd.cut(test_df['Basic_Demos-Age'], bins=[0, 18, 30, 40, np.inf], labels=['Child', 'Young Adult', 'Adult', 'Senior'])\n\n\ntrain_df[['Physical-BMI', 'BMI_Category', 'Basic_Demos-Age', 'Age_Group']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:46.565472Z","iopub.execute_input":"2024-12-04T07:49:46.566348Z","iopub.status.idle":"2024-12-04T07:49:46.589971Z","shell.execute_reply.started":"2024-12-04T07:49:46.566305Z","shell.execute_reply":"2024-12-04T07:49:46.588815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**4. Scaling and Encoding**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:49.026894Z","iopub.execute_input":"2024-12-04T07:49:49.027260Z","iopub.status.idle":"2024-12-04T07:49:49.032395Z","shell.execute_reply.started":"2024-12-04T07:49:49.027228Z","shell.execute_reply":"2024-12-04T07:49:49.031129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:50.459632Z","iopub.execute_input":"2024-12-04T07:49:50.459987Z","iopub.status.idle":"2024-12-04T07:49:50.465197Z","shell.execute_reply.started":"2024-12-04T07:49:50.459957Z","shell.execute_reply":"2024-12-04T07:49:50.464113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Editing:- try deleting this one during the final submission and then try to run the code\nencoder = OneHotEncoder(drop='first', sparse=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:51.835641Z","iopub.execute_input":"2024-12-04T07:49:51.836034Z","iopub.status.idle":"2024-12-04T07:49:51.840914Z","shell.execute_reply.started":"2024-12-04T07:49:51.835998Z","shell.execute_reply":"2024-12-04T07:49:51.839743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = OneHotEncoder(drop='first', sparse_output=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:53.177601Z","iopub.execute_input":"2024-12-04T07:49:53.177975Z","iopub.status.idle":"2024-12-04T07:49:53.182998Z","shell.execute_reply.started":"2024-12-04T07:49:53.177944Z","shell.execute_reply":"2024-12-04T07:49:53.181957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', scaler, numeric_features),\n        ('cat', encoder, categorical_features)\n    ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:54.743574Z","iopub.execute_input":"2024-12-04T07:49:54.743967Z","iopub.status.idle":"2024-12-04T07:49:54.749684Z","shell.execute_reply.started":"2024-12-04T07:49:54.743934Z","shell.execute_reply":"2024-12-04T07:49:54.748348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_processed = preprocessor.fit_transform(train_df)\nX_test_processed = preprocessor.transform(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:56.364943Z","iopub.execute_input":"2024-12-04T07:49:56.365336Z","iopub.status.idle":"2024-12-04T07:49:56.395142Z","shell.execute_reply.started":"2024-12-04T07:49:56.365302Z","shell.execute_reply":"2024-12-04T07:49:56.393960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder_fitted = preprocessor.named_transformers_['cat']\nprocessed_columns = numeric_features + list(encoder_fitted.get_feature_names_out(categorical_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:58.026987Z","iopub.execute_input":"2024-12-04T07:49:58.027377Z","iopub.status.idle":"2024-12-04T07:49:58.033233Z","shell.execute_reply.started":"2024-12-04T07:49:58.027340Z","shell.execute_reply":"2024-12-04T07:49:58.031750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_df = pd.DataFrame(X_train_processed, columns=processed_columns)\nX_test_df = pd.DataFrame(X_test_processed, columns=processed_columns)\n\nX_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:49:59.810910Z","iopub.execute_input":"2024-12-04T07:49:59.811307Z","iopub.status.idle":"2024-12-04T07:49:59.833904Z","shell.execute_reply.started":"2024-12-04T07:49:59.811267Z","shell.execute_reply":"2024-12-04T07:49:59.832776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Editing:- these codechunk gave us errors so delete this one while editing \n#processed_columns = numeric_features + list(encoder.get_feature_names_out(categorical_features))\n#X_train_df = pd.DataFrame(X_train_processed, columns=processed_columns)\n#X_test_df = pd.DataFrame(X_test_processed, columns=processed_columns)\n\n#X_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T20:43:29.205562Z","iopub.execute_input":"2024-12-01T20:43:29.206439Z","iopub.status.idle":"2024-12-01T20:43:29.211858Z","shell.execute_reply.started":"2024-12-01T20:43:29.206382Z","shell.execute_reply":"2024-12-01T20:43:29.210549Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**5. Outlier Detection**","metadata":{}},{"cell_type":"code","source":"def detect_outliers(df, features):\n    outlier_indices = []\n    for feature in features:\n        Q1 = df[feature].quantile(0.25)\n        Q3 = df[feature].quantile(0.75)\n        IQR = Q3 - Q1\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n        outliers = df[(df[feature] < lower_bound) | (df[feature] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n    return set(outlier_indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:03.309440Z","iopub.execute_input":"2024-12-04T07:50:03.309861Z","iopub.status.idle":"2024-12-04T07:50:03.317003Z","shell.execute_reply.started":"2024-12-04T07:50:03.309823Z","shell.execute_reply":"2024-12-04T07:50:03.315643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outliers = detect_outliers(train_df, numeric_features)\ntrain_df_cleaned = train_df.drop(index=outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:05.568495Z","iopub.execute_input":"2024-12-04T07:50:05.568916Z","iopub.status.idle":"2024-12-04T07:50:05.594694Z","shell.execute_reply.started":"2024-12-04T07:50:05.568882Z","shell.execute_reply":"2024-12-04T07:50:05.593736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Outliers detected: {len(outliers)}')\ntrain_df_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:07.403227Z","iopub.execute_input":"2024-12-04T07:50:07.403682Z","iopub.status.idle":"2024-12-04T07:50:07.427170Z","shell.execute_reply.started":"2024-12-04T07:50:07.403635Z","shell.execute_reply":"2024-12-04T07:50:07.426089Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**6. Data Visualization**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:11.126551Z","iopub.execute_input":"2024-12-04T07:50:11.126948Z","iopub.status.idle":"2024-12-04T07:50:11.132481Z","shell.execute_reply.started":"2024-12-04T07:50:11.126915Z","shell.execute_reply":"2024-12-04T07:50:11.131142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x='sii', data=train_df)\nplt.title('Distribution of Target Variable: sii')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:12.649649Z","iopub.execute_input":"2024-12-04T07:50:12.650037Z","iopub.status.idle":"2024-12-04T07:50:12.872926Z","shell.execute_reply.started":"2024-12-04T07:50:12.650004Z","shell.execute_reply":"2024-12-04T07:50:12.871884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'CGAS-CGAS_Score']].hist(bins=20, figsize=(10, 6))\nplt.suptitle('Distribution of Numeric Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:15.829107Z","iopub.execute_input":"2024-12-04T07:50:15.829795Z","iopub.status.idle":"2024-12-04T07:50:16.694240Z","shell.execute_reply.started":"2024-12-04T07:50:15.829753Z","shell.execute_reply":"2024-12-04T07:50:16.693158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = train_df[numeric_features].corr()\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title('Correlation Heatmap of Numeric Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:19.224138Z","iopub.execute_input":"2024-12-04T07:50:19.224584Z","iopub.status.idle":"2024-12-04T07:50:19.703237Z","shell.execute_reply.started":"2024-12-04T07:50:19.224511Z","shell.execute_reply":"2024-12-04T07:50:19.701962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**7. Training our model**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:22.384493Z","iopub.execute_input":"2024-12-04T07:50:22.384933Z","iopub.status.idle":"2024-12-04T07:50:22.506208Z","shell.execute_reply.started":"2024-12-04T07:50:22.384894Z","shell.execute_reply":"2024-12-04T07:50:22.505094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = train_df['sii']\n\nX_train_df = X_train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:23.717643Z","iopub.execute_input":"2024-12-04T07:50:23.718037Z","iopub.status.idle":"2024-12-04T07:50:23.723226Z","shell.execute_reply.started":"2024-12-04T07:50:23.718002Z","shell.execute_reply":"2024-12-04T07:50:23.722021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:25.524882Z","iopub.execute_input":"2024-12-04T07:50:25.525284Z","iopub.status.idle":"2024-12-04T07:50:25.530922Z","shell.execute_reply.started":"2024-12-04T07:50:25.525250Z","shell.execute_reply":"2024-12-04T07:50:25.529810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_train_df, y_train, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:27.184423Z","iopub.execute_input":"2024-12-04T07:50:27.184861Z","iopub.status.idle":"2024-12-04T07:50:27.194303Z","shell.execute_reply.started":"2024-12-04T07:50:27.184822Z","shell.execute_reply":"2024-12-04T07:50:27.193122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:28.660821Z","iopub.execute_input":"2024-12-04T07:50:28.661226Z","iopub.status.idle":"2024-12-04T07:50:28.669622Z","shell.execute_reply.started":"2024-12-04T07:50:28.661191Z","shell.execute_reply":"2024-12-04T07:50:28.668604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sample_submission_df.head())\n\nprint(set(train_df['id']).intersection(set(sample_submission_df['id'])))\n\ntrain_with_target_df = train_df.merge(sample_submission_df[['id', 'sii']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:30.187271Z","iopub.execute_input":"2024-12-04T07:50:30.187655Z","iopub.status.idle":"2024-12-04T07:50:30.207416Z","shell.execute_reply.started":"2024-12-04T07:50:30.187621Z","shell.execute_reply":"2024-12-04T07:50:30.205747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_with_target_df.drop(['id', 'sii'], axis=1)  # Features\ny_train = train_with_target_df['sii']\n\nmodel = LogisticRegression(random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:32.590264Z","iopub.execute_input":"2024-12-04T07:50:32.590692Z","iopub.status.idle":"2024-12-04T07:50:32.597915Z","shell.execute_reply.started":"2024-12-04T07:50:32.590651Z","shell.execute_reply":"2024-12-04T07:50:32.596740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:50:34.357370Z","iopub.execute_input":"2024-12-04T07:50:34.358323Z","iopub.status.idle":"2024-12-04T07:50:34.583245Z","shell.execute_reply.started":"2024-12-04T07:50:34.358277Z","shell.execute_reply":"2024-12-04T07:50:34.581733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:24.499084Z","iopub.execute_input":"2024-12-04T07:51:24.499493Z","iopub.status.idle":"2024-12-04T07:51:24.504475Z","shell.execute_reply.started":"2024-12-04T07:51:24.499456Z","shell.execute_reply":"2024-12-04T07:51:24.503137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sample_submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:35.712399Z","iopub.execute_input":"2024-12-04T07:51:35.712844Z","iopub.status.idle":"2024-12-04T07:51:35.721254Z","shell.execute_reply.started":"2024-12-04T07:51:35.712799Z","shell.execute_reply":"2024-12-04T07:51:35.720099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(set(train_df['id']).intersection(set(sample_submission_df['id'])))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:41.348857Z","iopub.execute_input":"2024-12-04T07:51:41.349250Z","iopub.status.idle":"2024-12-04T07:51:41.355847Z","shell.execute_reply.started":"2024-12-04T07:51:41.349215Z","shell.execute_reply":"2024-12-04T07:51:41.354552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_target_df = train_df.merge(sample_submission_df[['id', 'sii']])\n\nprint(train_with_target_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:44.305942Z","iopub.execute_input":"2024-12-04T07:51:44.306337Z","iopub.status.idle":"2024-12-04T07:51:44.329171Z","shell.execute_reply.started":"2024-12-04T07:51:44.306302Z","shell.execute_reply":"2024-12-04T07:51:44.328072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_with_target_df.drop(['id', 'sii'], axis=1)  # Features\ny_train = train_with_target_df['sii'] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:46.702961Z","iopub.execute_input":"2024-12-04T07:51:46.704650Z","iopub.status.idle":"2024-12-04T07:51:46.711941Z","shell.execute_reply.started":"2024-12-04T07:51:46.704492Z","shell.execute_reply":"2024-12-04T07:51:46.710480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import shuffle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:48.319379Z","iopub.execute_input":"2024-12-04T07:51:48.319759Z","iopub.status.idle":"2024-12-04T07:51:48.325249Z","shell.execute_reply.started":"2024-12-04T07:51:48.319724Z","shell.execute_reply":"2024-12-04T07:51:48.324037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train_bin = []\nfor label in y_train:\n    if label == 0:\n        y_train_bin.append(0)\n    else:\n        y_train_bin.append(1)\ny_train_bin = np.array(y_train_bin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:50.365066Z","iopub.execute_input":"2024-12-04T07:51:50.365862Z","iopub.status.idle":"2024-12-04T07:51:50.374010Z","shell.execute_reply.started":"2024-12-04T07:51:50.365792Z","shell.execute_reply":"2024-12-04T07:51:50.372256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LogisticRegression(random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:52.229817Z","iopub.execute_input":"2024-12-04T07:51:52.230222Z","iopub.status.idle":"2024-12-04T07:51:52.236310Z","shell.execute_reply.started":"2024-12-04T07:51:52.230185Z","shell.execute_reply":"2024-12-04T07:51:52.235136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X_train, y_train_bin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:53.506095Z","iopub.execute_input":"2024-12-04T07:51:53.506489Z","iopub.status.idle":"2024-12-04T07:51:53.555392Z","shell.execute_reply.started":"2024-12-04T07:51:53.506456Z","shell.execute_reply":"2024-12-04T07:51:53.553845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:56.912646Z","iopub.execute_input":"2024-12-04T07:51:56.913062Z","iopub.status.idle":"2024-12-04T07:51:56.970696Z","shell.execute_reply.started":"2024-12-04T07:51:56.913026Z","shell.execute_reply":"2024-12-04T07:51:56.969503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:58.312049Z","iopub.execute_input":"2024-12-04T07:51:58.312469Z","iopub.status.idle":"2024-12-04T07:51:58.317908Z","shell.execute_reply.started":"2024-12-04T07:51:58.312431Z","shell.execute_reply":"2024-12-04T07:51:58.316836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_encoder = LabelEncoder()\ny_train = label_encoder.fit_transform(y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:51:59.592977Z","iopub.execute_input":"2024-12-04T07:51:59.593358Z","iopub.status.idle":"2024-12-04T07:51:59.598967Z","shell.execute_reply.started":"2024-12-04T07:51:59.593324Z","shell.execute_reply":"2024-12-04T07:51:59.597770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Columns in X_train:\", X_train.columns)\nprint(\"Columns in X_val:\", X_val.columns)\nprint(\"Columns missing in X_val:\", set(X_train.columns) - set(X_val.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:01.047815Z","iopub.execute_input":"2024-12-04T07:52:01.048243Z","iopub.status.idle":"2024-12-04T07:52:01.054595Z","shell.execute_reply.started":"2024-12-04T07:52:01.048206Z","shell.execute_reply":"2024-12-04T07:52:01.053430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_val = X_val.reindex(columns=X_train.columns, fill_value=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:04.579658Z","iopub.execute_input":"2024-12-04T07:52:04.580155Z","iopub.status.idle":"2024-12-04T07:52:04.586416Z","shell.execute_reply.started":"2024-12-04T07:52:04.580100Z","shell.execute_reply":"2024-12-04T07:52:04.585260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.columns)\nprint(X_val.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:07.292491Z","iopub.execute_input":"2024-12-04T07:52:07.292926Z","iopub.status.idle":"2024-12-04T07:52:07.299549Z","shell.execute_reply.started":"2024-12-04T07:52:07.292891Z","shell.execute_reply":"2024-12-04T07:52:07.298338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.columns = X_train.columns.str.strip()\nX_val.columns = X_val.columns.str.strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:10.216151Z","iopub.execute_input":"2024-12-04T07:52:10.216584Z","iopub.status.idle":"2024-12-04T07:52:10.223281Z","shell.execute_reply.started":"2024-12-04T07:52:10.216510Z","shell.execute_reply":"2024-12-04T07:52:10.222074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', StandardScaler(), ['num_col1']),\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:11.763992Z","iopub.execute_input":"2024-12-04T07:52:11.764390Z","iopub.status.idle":"2024-12-04T07:52:11.769721Z","shell.execute_reply.started":"2024-12-04T07:52:11.764354Z","shell.execute_reply":"2024-12-04T07:52:11.768484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', StandardScaler(), [0]),  # Column index\n        # other transformers\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:13.469412Z","iopub.execute_input":"2024-12-04T07:52:13.469829Z","iopub.status.idle":"2024-12-04T07:52:13.475405Z","shell.execute_reply.started":"2024-12-04T07:52:13.469793Z","shell.execute_reply":"2024-12-04T07:52:13.474269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"X_train columns: \", X_train.columns)\nprint(\"X_val columns: \", X_val.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:15.048618Z","iopub.execute_input":"2024-12-04T07:52:15.049017Z","iopub.status.idle":"2024-12-04T07:52:15.056074Z","shell.execute_reply.started":"2024-12-04T07:52:15.048981Z","shell.execute_reply":"2024-12-04T07:52:15.054878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Hyperparameter tuning**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:18.203411Z","iopub.execute_input":"2024-12-04T07:52:18.203841Z","iopub.status.idle":"2024-12-04T07:52:18.208890Z","shell.execute_reply.started":"2024-12-04T07:52:18.203804Z","shell.execute_reply":"2024-12-04T07:52:18.207764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LogisticRegression(max_iter=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:19.645313Z","iopub.execute_input":"2024-12-04T07:52:19.645714Z","iopub.status.idle":"2024-12-04T07:52:19.651327Z","shell.execute_reply.started":"2024-12-04T07:52:19.645678Z","shell.execute_reply":"2024-12-04T07:52:19.650173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'C': [0.01, 0.1, 1, 10, 100],  # Regularization strength\n    'solver': ['liblinear', 'lbfgs'],  # Optimization algorithms\n    'penalty': ['l2', 'none']         # Regularization types\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:21.129309Z","iopub.execute_input":"2024-12-04T07:52:21.129777Z","iopub.status.idle":"2024-12-04T07:52:21.135681Z","shell.execute_reply.started":"2024-12-04T07:52:21.129736Z","shell.execute_reply":"2024-12-04T07:52:21.134381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_search = GridSearchCV(estimator=model, param_grid=param_grid, scoring='accuracy', cv=5, verbose=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:22.800619Z","iopub.execute_input":"2024-12-04T07:52:22.801016Z","iopub.status.idle":"2024-12-04T07:52:22.806430Z","shell.execute_reply.started":"2024-12-04T07:52:22.800981Z","shell.execute_reply":"2024-12-04T07:52:22.805264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_search.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:24.306054Z","iopub.execute_input":"2024-12-04T07:52:24.306454Z","iopub.status.idle":"2024-12-04T07:52:24.721144Z","shell.execute_reply.started":"2024-12-04T07:52:24.306417Z","shell.execute_reply":"2024-12-04T07:52:24.719603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Best Parameters:\", grid_search.best_params_)\nprint(\"Best Cross-Validation Score:\", grid_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:27.743473Z","iopub.execute_input":"2024-12-04T07:52:27.744620Z","iopub.status.idle":"2024-12-04T07:52:27.771250Z","shell.execute_reply.started":"2024-12-04T07:52:27.744554Z","shell.execute_reply":"2024-12-04T07:52:27.769791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = grid_search.best_estimator_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:30.660746Z","iopub.execute_input":"2024-12-04T07:52:30.661138Z","iopub.status.idle":"2024-12-04T07:52:30.687869Z","shell.execute_reply.started":"2024-12-04T07:52:30.661100Z","shell.execute_reply":"2024-12-04T07:52:30.686209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_val_pred = best_model.predict(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:33.319239Z","iopub.execute_input":"2024-12-04T07:52:33.319671Z","iopub.status.idle":"2024-12-04T07:52:33.347654Z","shell.execute_reply.started":"2024-12-04T07:52:33.319633Z","shell.execute_reply":"2024-12-04T07:52:33.346088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_accuracy = accuracy_score(y_val, y_val_pred)\nprint(\"Validation Accuracy with Best Parameters:\", val_accuracy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:35.338989Z","iopub.execute_input":"2024-12-04T07:52:35.339353Z","iopub.status.idle":"2024-12-04T07:52:35.366336Z","shell.execute_reply.started":"2024-12-04T07:52:35.339322Z","shell.execute_reply":"2024-12-04T07:52:35.364693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = model.predict(X_test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:37.213083Z","iopub.execute_input":"2024-12-04T07:52:37.213435Z","iopub.status.idle":"2024-12-04T07:52:37.368954Z","shell.execute_reply.started":"2024-12-04T07:52:37.213403Z","shell.execute_reply":"2024-12-04T07:52:37.367046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\nprint(\"Classification Report:\\n\", classification_report(y_val, y_val_pred))\nprint(\"Confusion Matrix:\\n\", confusion_matrix(y_val, y_val_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:40.953290Z","iopub.execute_input":"2024-12-04T07:52:40.953711Z","iopub.status.idle":"2024-12-04T07:52:40.980087Z","shell.execute_reply.started":"2024-12-04T07:52:40.953672Z","shell.execute_reply":"2024-12-04T07:52:40.978681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'C': [0.01, 0.1, 1, 10, 100],\n    'solver': ['liblinear', 'lbfgs', 'sag', 'saga'],\n    'penalty': ['l2', 'none']\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:42.616065Z","iopub.execute_input":"2024-12-04T07:52:42.616438Z","iopub.status.idle":"2024-12-04T07:52:42.621951Z","shell.execute_reply.started":"2024-12-04T07:52:42.616408Z","shell.execute_reply":"2024-12-04T07:52:42.620747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:44.370044Z","iopub.execute_input":"2024-12-04T07:52:44.370416Z","iopub.status.idle":"2024-12-04T07:52:44.375841Z","shell.execute_reply.started":"2024-12-04T07:52:44.370383Z","shell.execute_reply":"2024-12-04T07:52:44.374468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search = RandomizedSearchCV(\n    estimator=model,\n    param_distributions=param_grid,\n    n_iter=20,  # Number of parameter settings sampled\n    scoring='accuracy',\n    cv=5,\n    verbose=2,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:45.728684Z","iopub.execute_input":"2024-12-04T07:52:45.729074Z","iopub.status.idle":"2024-12-04T07:52:45.734805Z","shell.execute_reply.started":"2024-12-04T07:52:45.729042Z","shell.execute_reply":"2024-12-04T07:52:45.733662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:47.630392Z","iopub.execute_input":"2024-12-04T07:52:47.630772Z","iopub.status.idle":"2024-12-04T07:52:47.841512Z","shell.execute_reply.started":"2024-12-04T07:52:47.630738Z","shell.execute_reply":"2024-12-04T07:52:47.840004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Best Parameters (Random Search):\", random_search.best_params_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:51.239903Z","iopub.execute_input":"2024-12-04T07:52:51.240290Z","iopub.status.idle":"2024-12-04T07:52:51.267513Z","shell.execute_reply.started":"2024-12-04T07:52:51.240257Z","shell.execute_reply":"2024-12-04T07:52:51.265784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = pd.DataFrame(grid_search.cv_results_)\nprint(results[['mean_test_score', 'param_C', 'param_solver', 'param_penalty']])\n\n# Plot mean test scores for a quick visual\nsns.lineplot(data=results, x='param_C', y='mean_test_score', hue='param_solver')\nplt.title(\"Hyperparameter Tuning Results\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T07:52:52.664172Z","iopub.execute_input":"2024-12-04T07:52:52.664577Z","iopub.status.idle":"2024-12-04T07:52:52.693182Z","shell.execute_reply.started":"2024-12-04T07:52:52.664541Z","shell.execute_reply":"2024-12-04T07:52:52.691646Z"}},"outputs":[],"execution_count":null}]}