{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9977428,"sourceType":"datasetVersion","datasetId":6139070},{"sourceId":9985936,"sourceType":"datasetVersion","datasetId":6145342}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nbase_path = '/kaggle/input/cleaneddata/cleaned_train_copy.csv'\n\nimport pandas as pd\nimport numpy as np\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom imblearn.combine import SMOTETomek\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    accuracy_score,\n    classification_report,\n    confusion_matrix,\n    roc_auc_score,\n    roc_curve,\n    precision_recall_curve\n)\nfrom sklearn.svm import SVC\n# from sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.decomposition import PCA\nfrom imblearn.over_sampling import SMOTE\nimport seaborn as sns\nfrom sklearn.preprocessing import label_binarize\n\ndataset = pd.read_csv(base_path)\ndataset.head(5)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:09.247234Z","iopub.execute_input":"2024-12-06T04:29:09.248489Z","iopub.status.idle":"2024-12-06T04:29:10.299679Z","shell.execute_reply.started":"2024-12-06T04:29:09.248377Z","shell.execute_reply":"2024-12-06T04:29:10.298681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.drop(['age_group','id'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:10.301574Z","iopub.execute_input":"2024-12-06T04:29:10.302016Z","iopub.status.idle":"2024-12-06T04:29:10.313908Z","shell.execute_reply.started":"2024-12-06T04:29:10.301971Z","shell.execute_reply":"2024-12-06T04:29:10.312831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y = dataset.drop(columns=['sii']), dataset['sii']\nfeature_variances = np.var(X, axis=0)\nprint(f\"Feature Variances:\\n{feature_variances}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:10.315106Z","iopub.execute_input":"2024-12-06T04:29:10.315543Z","iopub.status.idle":"2024-12-06T04:29:10.334821Z","shell.execute_reply.started":"2024-12-06T04:29:10.315469Z","shell.execute_reply":"2024-12-06T04:29:10.333550Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"High Variance Features:\r\n\r\nSDS-SDS_Total_T (159.71): This feature has the highest variance and contributes significantly to the dataset's variability.\r\nCGAS_Score (118.72): Another highly variable feature that might be essential for modeling.\r\nPhysical-BMI (22.18): Moderate variance, still meaningful.\r\nModerate Variance Features:\r\n\r\nBasic_Demos-Age (11.71): Captures decent variability, likely important.\r\nPreInt_EduHx-computerinternet_hoursday (1.16): Lower but non-negligible variance.\r\nBIA_Activity_Level (1.06): Just above the threshold for near-zero variance.\r\nLow Variance Features:\r\n\r\nBasic_Demos-Sex (0.23): Minimal variability; likely a categorical or binary feature.\r\nFitness_Combined_Score (0.059): Very low variance; possibly not useful.\r\nPhysical_Composite_Index (0.00027): Extremely low variance, likely constant or nearly constant.","metadata":{}},{"cell_type":"code","source":"\nvariance_threshold = 0.1\n\nplt.figure(figsize=(10, 6))\nplt.bar(feature_variances.index, feature_variances.values)\nplt.axhline(y=variance_threshold, color='r', linestyle='--', label=\"Variance Threshold\")\nplt.xticks(rotation=45, ha=\"right\")\nplt.title(\"Feature Variance Distribution\")\nplt.ylabel(\"Variance\")\nplt.legend()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:10.337629Z","iopub.execute_input":"2024-12-06T04:29:10.338120Z","iopub.status.idle":"2024-12-06T04:29:11.031349Z","shell.execute_reply.started":"2024-12-06T04:29:10.338066Z","shell.execute_reply":"2024-12-06T04:29:11.029962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = feature_variances[feature_variances > 0.1].index\n\n# Filter the data for high-variance features\nX_high_variance = X[selected_features]\n\nprint(f\"Selected High Variance Features:\\n{selected_features}\\n\")\nprint(f\"Reduced Feature Set Shape: {X_high_variance.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:11.034857Z","iopub.execute_input":"2024-12-06T04:29:11.035386Z","iopub.status.idle":"2024-12-06T04:29:11.045496Z","shell.execute_reply.started":"2024-12-06T04:29:11.035307Z","shell.execute_reply":"2024-12-06T04:29:11.044341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PCA Analysis\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_high_variance)\npca = PCA(n_components=3)\nX_pca = pca.fit_transform(X_scaled)\npca_2d = PCA(n_components=2)\nX_pca_2d = pca_2d.fit_transform(X_scaled)\n\nexplained_variance = pca.explained_variance_ratio_\nprint(f\"Explained Variance Ratio (PCA): {explained_variance}\")\nprint(f\"Cumulative Explained Variance: {np.cumsum(explained_variance)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:11.047042Z","iopub.execute_input":"2024-12-06T04:29:11.047486Z","iopub.status.idle":"2024-12-06T04:29:11.086484Z","shell.execute_reply.started":"2024-12-06T04:29:11.047440Z","shell.execute_reply":"2024-12-06T04:29:11.085399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize PCA components\nplt.figure(figsize=(8, 6))\nplt.scatter(X_pca[:, 0], X_pca[:, 1], c=y, cmap=\"viridis\", s=30, edgecolor=\"k\", alpha=0.7)\nplt.title(\"PCA Visualization (PC1 vs PC2)\")\nplt.xlabel(\"PC1\")\nplt.ylabel(\"PC2\")\nplt.colorbar(label=\"Target Variable\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:11.087694Z","iopub.execute_input":"2024-12-06T04:29:11.088138Z","iopub.status.idle":"2024-12-06T04:29:11.548815Z","shell.execute_reply.started":"2024-12-06T04:29:11.088089Z","shell.execute_reply":"2024-12-06T04:29:11.547394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3D Plot of PCA\nfrom mpl_toolkits.mplot3d import Axes3D\n\nfig = plt.figure(figsize=(10, 8))\nax = fig.add_subplot(111, projection=\"3d\")\nsc = ax.scatter(X_pca[:, 0], X_pca[:, 1], X_pca[:, 2], c=y, cmap=\"viridis\", s=50, alpha=0.7)\nplt.title(\"3D Plot of PCA Components (PC1, PC2, PC3)\")\nplt.colorbar(sc, label=\"Target Variable\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:11.550517Z","iopub.execute_input":"2024-12-06T04:29:11.550975Z","iopub.status.idle":"2024-12-06T04:29:12.009973Z","shell.execute_reply.started":"2024-12-06T04:29:11.550929Z","shell.execute_reply":"2024-12-06T04:29:12.008688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scree Plot for individual explained variance\nplt.figure(figsize=(8, 6))\nplt.bar(range(1, len(explained_variance) + 1), explained_variance, color=\"skyblue\", edgecolor=\"black\")\nplt.title(\"Scree Plot (Explained Variance by Each Principal Component)\")\nplt.xlabel(\"Principal Component\")\nplt.ylabel(\"Explained Variance Ratio\")\nplt.xticks(range(1, len(explained_variance) + 1))\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.7)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:12.011566Z","iopub.execute_input":"2024-12-06T04:29:12.012054Z","iopub.status.idle":"2024-12-06T04:29:12.270871Z","shell.execute_reply.started":"2024-12-06T04:29:12.012001Z","shell.execute_reply":"2024-12-06T04:29:12.269522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.naive_bayes import GaussianNB\n\n# Naive Bayes Baseline to check how model is actually performing\nX_train, X_test, y_train, y_test = train_test_split(X_pca_2d, y, test_size=0.3, random_state=42, stratify=y)\nnb_model = GaussianNB()\nnb_model.fit(X_train, y_train)\ny_pred_nb = nb_model.predict(X_test)\n\nprint(\"\\nNaive Bayes Classification Report:\")\nprint(classification_report(y_test, y_pred_nb))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:12.272723Z","iopub.execute_input":"2024-12-06T04:29:12.273181Z","iopub.status.idle":"2024-12-06T04:29:12.302242Z","shell.execute_reply.started":"2024-12-06T04:29:12.273129Z","shell.execute_reply":"2024-12-06T04:29:12.300833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I will try out two SVM models, one without resampling and one with","metadata":{}},{"cell_type":"code","source":"# Common SVM hyperparameters for both models\nparam_grid = {\n    'C': [1],\n    'gamma': [0.7],\n    'kernel': ['rbf']\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:12.303843Z","iopub.execute_input":"2024-12-06T04:29:12.304200Z","iopub.status.idle":"2024-12-06T04:29:12.309287Z","shell.execute_reply.started":"2024-12-06T04:29:12.304163Z","shell.execute_reply":"2024-12-06T04:29:12.308101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Model 1: Without Resampling\nprint(\"Training SVM without resampling...\")\ngrid_search_no_resampling = GridSearchCV(\n    SVC(probability=True, class_weight='balanced', random_state=42),\n    param_grid, cv=3, scoring='f1_weighted', n_jobs=-1\n)\ngrid_search_no_resampling.fit(X_train, y_train)\nbest_params_no_resampling = grid_search_no_resampling.best_params_\nprint(\"Best Parameters (No Resampling):\", best_params_no_resampling)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:12.310717Z","iopub.execute_input":"2024-12-06T04:29:12.311090Z","iopub.status.idle":"2024-12-06T04:29:16.675736Z","shell.execute_reply.started":"2024-12-06T04:29:12.311057Z","shell.execute_reply":"2024-12-06T04:29:16.674362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate on the test set\ny_pred_no_resampling = grid_search_no_resampling.best_estimator_.predict(X_test)\nprint(\"\\nClassification Report (No Resampling):\")\nprint(classification_report(y_test, y_pred_no_resampling))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:16.677438Z","iopub.execute_input":"2024-12-06T04:29:16.677862Z","iopub.status.idle":"2024-12-06T04:29:16.759890Z","shell.execute_reply.started":"2024-12-06T04:29:16.677824Z","shell.execute_reply":"2024-12-06T04:29:16.758402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Model 2: With Resampling (SMOTE-Tomek)\nprint(\"\\nTraining SVM with resampling...\")\nsmote_tomek = SMOTETomek(random_state=42)\nX_resampled, y_resampled = smote_tomek.fit_resample(X_train, y_train)\n\ngrid_search_with_resampling = GridSearchCV(\n    SVC(probability=True, class_weight='balanced', random_state=42),\n    param_grid, cv=3, scoring='f1_weighted', n_jobs=-1\n)\ngrid_search_with_resampling.fit(X_resampled, y_resampled)\nbest_params_with_resampling = grid_search_with_resampling.best_params_\nprint(\"Best Parameters (With Resampling):\", best_params_with_resampling)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:16.764221Z","iopub.execute_input":"2024-12-06T04:29:16.764576Z","iopub.status.idle":"2024-12-06T04:29:21.516940Z","shell.execute_reply.started":"2024-12-06T04:29:16.764544Z","shell.execute_reply":"2024-12-06T04:29:21.515680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate on the test set\ny_pred_with_resampling = grid_search_with_resampling.best_estimator_.predict(X_test)\nprint(\"\\nClassification Report (With Resampling):\")\nprint(classification_report(y_test, y_pred_with_resampling))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:21.518169Z","iopub.execute_input":"2024-12-06T04:29:21.518511Z","iopub.status.idle":"2024-12-06T04:29:21.646511Z","shell.execute_reply.started":"2024-12-06T04:29:21.518477Z","shell.execute_reply":"2024-12-06T04:29:21.645308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hyperparameter Tuning for SVM\nparam_grid = {\n    'C': [1],\n    'gamma': [0.7],\n    'kernel': ['rbf'],\n    'class_weight': ['balanced']\n}\ngrid_search = GridSearchCV(SVC(probability=True, class_weight='balanced', random_state=42),\n                           param_grid, cv=3, scoring='f1_weighted', n_jobs=-1)\n\ngrid_search.fit(X_resampled, y_resampled)\nbest_params = grid_search.best_params_\nprint(\"Best Parameters:\", best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:21.647720Z","iopub.execute_input":"2024-12-06T04:29:21.648022Z","iopub.status.idle":"2024-12-06T04:29:25.718098Z","shell.execute_reply.started":"2024-12-06T04:29:21.647991Z","shell.execute_reply":"2024-12-06T04:29:25.716824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train SVM with best parameters\nsvm_model = grid_search.best_estimator_\nsvm_model.fit(X_resampled, y_resampled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:25.719738Z","iopub.execute_input":"2024-12-06T04:29:25.720188Z","iopub.status.idle":"2024-12-06T04:29:28.015436Z","shell.execute_reply.started":"2024-12-06T04:29:25.720141Z","shell.execute_reply":"2024-12-06T04:29:28.014266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Support Vectors\nsupport_vectors = svm_model.support_vectors_\nnum_support_vectors = len(support_vectors)\nsupport_vector_indices = svm_model.support_\nprint(f\"Number of Support Vectors: {num_support_vectors}\")\nprint(f\"Support Vector Indices: {support_vector_indices}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.017220Z","iopub.execute_input":"2024-12-06T04:29:28.017702Z","iopub.status.idle":"2024-12-06T04:29:28.025807Z","shell.execute_reply.started":"2024-12-06T04:29:28.017652Z","shell.execute_reply":"2024-12-06T04:29:28.024504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of Support Vectors: {svm_model.support_vectors_.shape[0]}\")\nprint(f\"Support Vectors per Class:\")\nfor i, count in enumerate(svm_model.n_support_):\n    print(f\"  Class {i}: {count} support vectors\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.026971Z","iopub.execute_input":"2024-12-06T04:29:28.027271Z","iopub.status.idle":"2024-12-06T04:29:28.041010Z","shell.execute_reply.started":"2024-12-06T04:29:28.027242Z","shell.execute_reply":"2024-12-06T04:29:28.039689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = svm_model.predict(X_test)\ny_prob = svm_model.predict_proba(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.042560Z","iopub.execute_input":"2024-12-06T04:29:28.043036Z","iopub.status.idle":"2024-12-06T04:29:28.279764Z","shell.execute_reply.started":"2024-12-06T04:29:28.042989Z","shell.execute_reply":"2024-12-06T04:29:28.278858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluating Precision-Recall and ROC Curves for each class\nplt.figure(figsize=(10, 8))\ny_test_binarized = label_binarize(y_test, classes=np.unique(y))\nfor i, class_label in enumerate(np.unique(y)):\n    precision, recall, _ = precision_recall_curve(y_test_binarized[:, i], y_prob[:, i])\n    plt.plot(recall, precision, label=f\"Class {class_label}\")\nplt.title(\"Precision-Recall Curves\")\nplt.xlabel(\"Recall\")\nplt.ylabel(\"Precision\")\nplt.legend()\nplt.show()\n\nplt.figure(figsize=(10, 8))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.281098Z","iopub.execute_input":"2024-12-06T04:29:28.281450Z","iopub.status.idle":"2024-12-06T04:29:28.620019Z","shell.execute_reply.started":"2024-12-06T04:29:28.281417Z","shell.execute_reply":"2024-12-06T04:29:28.618830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix Visualization\nplt.figure(figsize=(8, 6))\nsns.heatmap(confusion_matrix(y_test, y_pred), annot=True, fmt='d', cmap=\"coolwarm\", xticklabels=np.unique(y), yticklabels=np.unique(y))\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.621460Z","iopub.execute_input":"2024-12-06T04:29:28.621910Z","iopub.status.idle":"2024-12-06T04:29:28.960874Z","shell.execute_reply.started":"2024-12-06T04:29:28.621861Z","shell.execute_reply":"2024-12-06T04:29:28.959281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"roc_auc_scores = roc_auc_score(y_test_binarized, y_prob, average=None)\nfor i, class_label in enumerate(np.unique(y)):\n    fpr, tpr, _ = roc_curve(y_test_binarized[:, i], y_prob[:, i])\n    plt.plot(fpr, tpr, label=f\"Class {class_label} (AUC = {roc_auc_scores[i]:.2f})\")\nplt.plot([0, 1], [0, 1], 'k--')\nplt.title(\"ROC Curves\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.legend()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:28.962541Z","iopub.execute_input":"2024-12-06T04:29:28.962963Z","iopub.status.idle":"2024-12-06T04:29:29.280175Z","shell.execute_reply.started":"2024-12-06T04:29:28.962921Z","shell.execute_reply":"2024-12-06T04:29:29.279041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import (classification_report, confusion_matrix, \n                              roc_curve, roc_auc_score, precision_recall_curve, \n                              average_precision_score)\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.model_selection import GridSearchCV\n\n# Visualizing Decision Boundary for each kernel\ndef plot_decision_boundary(X, y, model, title=\"Decision Boundary\"):\n    # Create a mesh grid for plotting\n    h = .02  # Step size in the mesh\n    x_min, x_max = X[:, 0].min() - 1, X[:, 0].max() + 1\n    y_min, y_max = X[:, 1].min() - 1, X[:, 1].max() + 1\n    xx, yy = np.meshgrid(np.arange(x_min, x_max, h),\n                         np.arange(y_min, y_max, h))\n\n    # Predict labels for the entire mesh grid\n    Z = model.predict(np.c_[xx.ravel(), yy.ravel()])\n    Z = Z.reshape(xx.shape)\n\n    # Plot the contour and decision boundary\n    plt.contourf(xx, yy, Z, alpha=0.75, cmap='viridis')\n    plt.scatter(X[:, 0], X[:, 1], c=y, edgecolors='k', cmap='viridis')\n    plt.title(title)\n    plt.xlabel('PCA Component 1')\n    plt.ylabel('PCA Component 2')\n    plt.tight_layout()\n    plt.show()\n\n\n# Hyperparameter Tuning for each kernel\nkernels = ['linear', 'poly', 'rbf']\nresults = {}\n\n# Train and tune SVM for each kernel\nfor kernel in kernels:\n    print(f\"\\nTraining with {kernel} kernel...\")\n\n    # Define the param_grid specific to each kernel\n    if kernel == 'linear':\n        param_grid = {\n            'C': [1],\n            'kernel': ['linear'],\n            'class_weight': ['balanced']\n        }\n    elif kernel == 'poly':\n        param_grid = {\n            'C': [1],\n            'gamma': [0.7],\n            'kernel': ['poly'],\n            'class_weight': ['balanced']\n        }\n    elif kernel == 'rbf':\n        param_grid = {\n            'C': [1],\n            'gamma': [0.7],\n            'kernel': ['rbf'],\n            'class_weight': ['balanced']\n        }\n\n    # Perform Grid Search\n    grid_search = GridSearchCV(SVC(probability=True, random_state=42),\n                               param_grid, cv=3, scoring='f1_weighted', n_jobs=-1)\n    grid_search.fit(X_resampled, y_resampled)\n    \n    svm_model_test = grid_search.best_estimator_\n\n    # Predictions and Probabilities\n    y_pred = svm_model_test.predict(X_test)\n    y_prob = svm_model_test.predict_proba(X_test)\n    \n    # Classification Report\n    print(f\"\\n{kernel.upper()} Kernel Classification Report:\")\n    print(classification_report(y_test, y_pred))\n    \n    # Confusion Matrix\n    plt.figure(figsize=(8, 6))\n    cm = confusion_matrix(y_test, y_pred)\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\n    plt.title(f'{kernel.upper()} Kernel Confusion Matrix')\n    plt.ylabel('True Label')\n    plt.xlabel('Predicted Label')\n    plt.tight_layout()\n    plt.show()\n\n    # Precision-Recall Curve\n    plt.figure(figsize=(10, 8))\n    y_test_bin = label_binarize(y_test, classes=np.unique(y_test))\n    \n    for i in range(len(np.unique(y_test))):\n        precision, recall, _ = precision_recall_curve(y_test_bin[:, i], y_prob[:, i])\n        average_precision = average_precision_score(y_test_bin[:, i], y_prob[:, i])\n        plt.plot(recall, precision, label=f'Class {i} (AP = {average_precision:.2f})')\n    \n    plt.title(f'{kernel.upper()} Kernel Precision-Recall Curve')\n    plt.xlabel('Recall')\n    plt.ylabel('Precision')\n    plt.legend(loc='best')\n    plt.tight_layout()\n    plt.show()\n\n    # ROC Curve for Multiclass\n    plt.figure(figsize=(10, 8))\n    for i in range(len(np.unique(y_test))):\n        fpr, tpr, _ = roc_curve(y_test_bin[:, i], y_prob[:, i])\n        roc_auc = roc_auc_score(y_test_bin[:, i], y_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (AUC = {roc_auc:.2f})')\n    \n    plt.plot([0, 1], [0, 1], 'k--')\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title(f'{kernel.upper()} Kernel ROC Curve')\n    plt.legend(loc=\"lower right\")\n    plt.tight_layout()\n    plt.show()\n\n     # Support Vectors Info\n    print(\"\\nSVM Model Characteristics:\")\n    print(f\"Number of Support Vectors: {svm_model_test.support_vectors_.shape[0]}\")\n    print(f\"Support Vectors per Class:\")\n    for i, count in enumerate(svm_model_test.n_support_):\n        print(f\"  Class {i}: {count} support vectors\")\n\n    # Plot the decision boundary for the kernel\n    plot_decision_boundary(X_resampled, y_resampled, svm_model_test, title=f'{kernel.upper()} Kernel Decision Boundary')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:29:29.281774Z","iopub.execute_input":"2024-12-06T04:29:29.282184Z","iopub.status.idle":"2024-12-06T04:31:08.622512Z","shell.execute_reply.started":"2024-12-06T04:29:29.282137Z","shell.execute_reply":"2024-12-06T04:31:08.621398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Metrics\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred))\nprint(\"Confusion Matrix:\\n\", confusion_matrix(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:31:08.623926Z","iopub.execute_input":"2024-12-06T04:31:08.624254Z","iopub.status.idle":"2024-12-06T04:31:08.643038Z","shell.execute_reply.started":"2024-12-06T04:31:08.624223Z","shell.execute_reply":"2024-12-06T04:31:08.642113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n\npath_to_scaler = '/kaggle/working/svm_tabulardata_scaler.pkl'\npath_to_pca = '/kaggle/working/svm_tabulardata_pca.pkl'\npath_to_svm_model = '/kaggle/working/svm_tabulardata_model.pkl'\npath_to_selected_features = '/kaggle/working/svm_tabulardata_selected_features.pkl'\n\n\n# Save the trained model to the specified path\njoblib.dump(svm_model, path_to_svm_model)\njoblib.dump(scaler, path_to_scaler)  # Save scaler\njoblib.dump(pca_2d, path_to_pca)  # Save PCA\njoblib.dump(selected_features, path_to_selected_features) # save selected features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:31:08.644881Z","iopub.execute_input":"2024-12-06T04:31:08.645314Z","iopub.status.idle":"2024-12-06T04:31:08.661161Z","shell.execute_reply.started":"2024-12-06T04:31:08.645269Z","shell.execute_reply":"2024-12-06T04:31:08.660122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we will use this saved model to do the testing on our test dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport joblib\nfrom sklearn.metrics import confusion_matrix, classification_report\n\npath_to_scaler = '/kaggle/working/svm_tabulardata_scaler.pkl'\npath_to_pca = '/kaggle/working/svm_tabulardata_pca.pkl'\npath_to_svm_model = '/kaggle/working/svm_tabulardata_model.pkl'\npath_to_selected_features = '/kaggle/working/svm_tabulardata_selected_features.pkl'\n\n#Load Saved Components\nscaler = joblib.load(path_to_scaler)\npca_2d = joblib.load(path_to_pca)\nsvm_model = joblib.load(path_to_svm_model)\nselected_features = joblib.load(path_to_selected_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:31:08.662365Z","iopub.execute_input":"2024-12-06T04:31:08.662776Z","iopub.status.idle":"2024-12-06T04:31:08.678723Z","shell.execute_reply.started":"2024-12-06T04:31:08.662718Z","shell.execute_reply":"2024-12-06T04:31:08.677793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the new test dataset\ntest_data_path = '/kaggle/input/testdataset/cleaned_test.csv'\ntest_dataset = pd.read_csv(test_data_path)\n\n# Check the number of missing values per column\nmissing_values_count = test_dataset.isnull().sum()\n\n# Print out the count of missing values for each column\nprint(\"Missing values count per column:\")\nprint(missing_values_count)\n\ntotal_missing_values = missing_values_count.sum()\nprint(f\"Total number of missing values in the dataset: {total_missing_values}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:31:08.680470Z","iopub.execute_input":"2024-12-06T04:31:08.680941Z","iopub.status.idle":"2024-12-06T04:31:08.704359Z","shell.execute_reply.started":"2024-12-06T04:31:08.680893Z","shell.execute_reply":"2024-12-06T04:31:08.703339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply feature selection (using selected_features from the training phase)\nX_test_high_variance = test_dataset[selected_features]\n\n# Scale and Apply PCA (using the saved scaler and PCA)\nX_test_scaled = scaler.transform(X_test_high_variance)\nX_test_pca = pca_2d.transform(X_test_scaled)\n\n# Make Predictions\ny_pred = svm_model.predict(X_test_pca)\ny_pred_proba = svm_model.predict_proba(X_test_pca)\n\npredictions_df = pd.DataFrame({\n    'id': test_dataset['id'],  # Replace with actual id column\n    'sii': y_pred\n})\npredictions_df.to_csv('/kaggle/working/submission.csv', index=False)\n\npredictions_proba_df = pd.DataFrame(y_pred_proba, columns=[f'Class_{i}_Prob' for i in range(y_pred_proba.shape[1])])\npredictions_proba_df.to_csv('/kaggle/working/predictions_with_probabilities.csv', index=False)\n\n# Output the predictions\nprint(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T04:31:08.705654Z","iopub.execute_input":"2024-12-06T04:31:08.705986Z","iopub.status.idle":"2024-12-06T04:31:08.728353Z","shell.execute_reply.started":"2024-12-06T04:31:08.705953Z","shell.execute_reply":"2024-12-06T04:31:08.727161Z"}},"outputs":[],"execution_count":null}]}