{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport os\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nfrom scipy.optimize import minimize\nfrom xgboost import XGBRegressor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-28T17:50:57.761743Z","iopub.execute_input":"2024-09-28T17:50:57.763367Z","iopub.status.idle":"2024-09-28T17:50:57.771290Z","shell.execute_reply.started":"2024-09-28T17:50:57.763305Z","shell.execute_reply":"2024-09-28T17:50:57.770002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.773713Z","iopub.execute_input":"2024-09-28T17:50:57.774294Z","iopub.status.idle":"2024-09-28T17:50:57.852890Z","shell.execute_reply.started":"2024-09-28T17:50:57.774241Z","shell.execute_reply":"2024-09-28T17:50:57.851541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.854763Z","iopub.execute_input":"2024-09-28T17:50:57.855219Z","iopub.status.idle":"2024-09-28T17:50:57.891877Z","shell.execute_reply.started":"2024-09-28T17:50:57.855168Z","shell.execute_reply":"2024-09-28T17:50:57.890766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_train = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nseries_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.894459Z","iopub.execute_input":"2024-09-28T17:50:57.895157Z","iopub.status.idle":"2024-09-28T17:50:57.955554Z","shell.execute_reply.started":"2024-09-28T17:50:57.895108Z","shell.execute_reply":"2024-09-28T17:50:57.954219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.957117Z","iopub.execute_input":"2024-09-28T17:50:57.957601Z","iopub.status.idle":"2024-09-28T17:50:57.982079Z","shell.execute_reply.started":"2024-09-28T17:50:57.957550Z","shell.execute_reply":"2024-09-28T17:50:57.980684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.984058Z","iopub.execute_input":"2024-09-28T17:50:57.984552Z","iopub.status.idle":"2024-09-28T17:50:57.995249Z","shell.execute_reply.started":"2024-09-28T17:50:57.984499Z","shell.execute_reply":"2024-09-28T17:50:57.993516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:57.997591Z","iopub.execute_input":"2024-09-28T17:50:57.998273Z","iopub.status.idle":"2024-09-28T17:50:58.015392Z","shell.execute_reply.started":"2024-09-28T17:50:57.998214Z","shell.execute_reply":"2024-09-28T17:50:58.014037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.016790Z","iopub.execute_input":"2024-09-28T17:50:58.017129Z","iopub.status.idle":"2024-09-28T17:50:58.024363Z","shell.execute_reply.started":"2024-09-28T17:50:58.017095Z","shell.execute_reply":"2024-09-28T17:50:58.022818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cat = train[categorical]\ntest_cat = test[categorical]","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.028174Z","iopub.execute_input":"2024-09-28T17:50:58.028926Z","iopub.status.idle":"2024-09-28T17:50:58.038154Z","shell.execute_reply.started":"2024-09-28T17:50:58.028883Z","shell.execute_reply":"2024-09-28T17:50:58.036957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.042974Z","iopub.execute_input":"2024-09-28T17:50:58.043392Z","iopub.status.idle":"2024-09-28T17:50:58.064420Z","shell.execute_reply.started":"2024-09-28T17:50:58.043351Z","shell.execute_reply":"2024-09-28T17:50:58.063140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nEksik değerler:\")\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.066446Z","iopub.execute_input":"2024-09-28T17:50:58.066943Z","iopub.status.idle":"2024-09-28T17:50:58.081834Z","shell.execute_reply.started":"2024-09-28T17:50:58.066890Z","shell.execute_reply":"2024-09-28T17:50:58.080251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_train = train.select_dtypes(include=['float', 'int'])\nnumeric_test = test.select_dtypes(include=['float', 'int'])\n\n# Count null values in each numeric column\nnull_train = numeric_train.isna().sum()\nnull_test = numeric_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.083591Z","iopub.execute_input":"2024-09-28T17:50:58.084206Z","iopub.status.idle":"2024-09-28T17:50:58.095359Z","shell.execute_reply.started":"2024-09-28T17:50:58.084152Z","shell.execute_reply":"2024-09-28T17:50:58.093836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_train\nnull_test\nfor column in numeric_train.columns:\n    mode_train = numeric_train[column].mode()[0] \n    numeric_train[column].fillna(mode_train, inplace=True)\n\nfor column in numeric_test.columns:\n    mode_test = numeric_test[column].mode()[0]\n    numeric_test[column].fillna(mode_test, inplace=True)\n\ntrain[numeric_train.columns] = numeric_train\ntest[numeric_test.columns] = numeric_test","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.097529Z","iopub.execute_input":"2024-09-28T17:50:58.098504Z","iopub.status.idle":"2024-09-28T17:50:58.169635Z","shell.execute_reply.started":"2024-09-28T17:50:58.098447Z","shell.execute_reply":"2024-09-28T17:50:58.168473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\ndef plot_histograms(df, num_bins=30):\n    sns.set(style=\"whitegrid\")\n\n    # Get only the numeric columns for plotting histograms\n    numeric_columns = df.select_dtypes(include=['float64', 'int64']).columns\n    num_features = len(numeric_columns)\n\n    plt.figure(figsize=(15, num_features * 4))\n\n    for i, column in enumerate(numeric_columns):\n        plt.subplot(num_features, 1, i + 1)\n        sns.histplot(df[column].dropna(), bins=num_bins, kde=False, color='green')\n        plt.title(f'Histogram of {column}', fontsize=14)\n        plt.xlabel(column)\n        plt.ylabel('Count')\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.171125Z","iopub.execute_input":"2024-09-28T17:50:58.171649Z","iopub.status.idle":"2024-09-28T17:50:58.182549Z","shell.execute_reply.started":"2024-09-28T17:50:58.171604Z","shell.execute_reply":"2024-09-28T17:50:58.180935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_histograms(train)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:50:58.184167Z","iopub.execute_input":"2024-09-28T17:50:58.184641Z","iopub.status.idle":"2024-09-28T17:51:18.988047Z","shell.execute_reply.started":"2024-09-28T17:50:58.184591Z","shell.execute_reply":"2024-09-28T17:51:18.986606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[categorical] = train_cat.fillna('missing')\ntest[categorical] = test_cat.fillna('missing')\ntrain[categorical] = train_cat.astype('category')\ntest[categorical] = test_cat.astype('category')\n\nfrom sklearn.preprocessing import LabelEncoder\n\ndef label_encode_columns(dataset, categorical_columns):\n    label_encoders = {}\n    for col in categorical_columns:\n        le = LabelEncoder()\n        dataset[col] = le.fit_transform(dataset[col].astype(str))\n        label_encoders[col] = le \n    return dataset, label_encoders\n\n\ntrain_encoded, label_encoders_train = label_encode_columns(train, categorical)\ntest_encoded, label_encoders_test = label_encode_columns(test, categorical)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:18.989547Z","iopub.execute_input":"2024-09-28T17:51:18.989984Z","iopub.status.idle":"2024-09-28T17:51:19.043920Z","shell.execute_reply.started":"2024-09-28T17:51:18.989940Z","shell.execute_reply":"2024-09-28T17:51:19.042717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:19.045456Z","iopub.execute_input":"2024-09-28T17:51:19.045840Z","iopub.status.idle":"2024-09-28T17:51:19.066173Z","shell.execute_reply.started":"2024-09-28T17:51:19.045802Z","shell.execute_reply":"2024-09-28T17:51:19.064966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nMissing values:\")\ntrain.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:19.067560Z","iopub.execute_input":"2024-09-28T17:51:19.067941Z","iopub.status.idle":"2024-09-28T17:51:19.082414Z","shell.execute_reply.started":"2024-09-28T17:51:19.067904Z","shell.execute_reply":"2024-09-28T17:51:19.081174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install missingno","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:19.083645Z","iopub.execute_input":"2024-09-28T17:51:19.084000Z","iopub.status.idle":"2024-09-28T17:51:30.572533Z","shell.execute_reply.started":"2024-09-28T17:51:19.083964Z","shell.execute_reply":"2024-09-28T17:51:30.570875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as msno\n\nmsno.matrix(train)\nplt.title(\"Missing Data Matrix\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:30.574590Z","iopub.execute_input":"2024-09-28T17:51:30.575273Z","iopub.status.idle":"2024-09-28T17:51:31.141881Z","shell.execute_reply.started":"2024-09-28T17:51:30.575223Z","shell.execute_reply":"2024-09-28T17:51:31.140679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Descriptive statistics:\")\nprint(train.describe())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:31.143468Z","iopub.execute_input":"2024-09-28T17:51:31.143877Z","iopub.status.idle":"2024-09-28T17:51:31.269999Z","shell.execute_reply.started":"2024-09-28T17:51:31.143836Z","shell.execute_reply":"2024-09-28T17:51:31.268844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40, 20))\nmask = np.triu(np.ones_like(train.corr(), dtype=bool))\nsns.heatmap(train.corr(), mask=mask, annot=True, cmap=\"coolwarm\", linewidths=0.5, fmt=\".2f\", square=True)\nplt.title(\"Correlation Matrix\", fontsize=30)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:31.271398Z","iopub.execute_input":"2024-09-28T17:51:31.271742Z","iopub.status.idle":"2024-09-28T17:51:37.837708Z","shell.execute_reply.started":"2024-09-28T17:51:31.271707Z","shell.execute_reply":"2024-09-28T17:51:37.836494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install yellowbrick\nfrom yellowbrick.features import rank1d\n\nfig, ax = plt.subplots(figsize=(20, 15))\n\n# Visualize the rank1d plot\nvisualizer = rank1d(train, ax=None)\nvisualizer.fit(train)        \nvisualizer.transform(train)  \nvisualizer.show()            \n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:37.839496Z","iopub.execute_input":"2024-09-28T17:51:37.839861Z","iopub.status.idle":"2024-09-28T17:51:50.457158Z","shell.execute_reply.started":"2024-09-28T17:51:37.839826Z","shell.execute_reply":"2024-09-28T17:51:50.455825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from yellowbrick.features import rank2d\n\nfig, ax = plt.subplots(figsize=(20, 15))\nrank2d(train, algorithm='pearson');","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:50.459025Z","iopub.execute_input":"2024-09-28T17:51:50.459435Z","iopub.status.idle":"2024-09-28T17:51:52.068898Z","shell.execute_reply.started":"2024-09-28T17:51:50.459394Z","shell.execute_reply":"2024-09-28T17:51:52.067736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(20, 15))\nrank2d(train, algorithm='covariance');","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:52.075726Z","iopub.execute_input":"2024-09-28T17:51:52.076163Z","iopub.status.idle":"2024-09-28T17:51:53.671216Z","shell.execute_reply.started":"2024-09-28T17:51:52.076117Z","shell.execute_reply":"2024-09-28T17:51:53.669978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train.corr() \ncorr.style.background_gradient(cmap='coolwarm')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:53.672715Z","iopub.execute_input":"2024-09-28T17:51:53.673140Z","iopub.status.idle":"2024-09-28T17:51:53.949494Z","shell.execute_reply.started":"2024-09-28T17:51:53.673097Z","shell.execute_reply":"2024-09-28T17:51:53.948142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef correlated_columns(train, threshold_corr, target_col=None):\n    \"\"\"\n    Identifies columns that are highly correlated with each other based on the threshold.\n    \n    Parameters:\n    df (pd.DataFrame): The DataFrame containing the data.\n    threshold_corr (float): Correlation threshold for dropping features (between 0 and 1).\n    target_col (str, optional): Target column to exclude from dropping, if specified.\n    \n    Returns:\n    list: A list of highly correlated columns that can be dropped.\n    \"\"\"\n    corr_matrix = train.corr().abs()  # Get the absolute correlation matrix\n    upper_triangle = np.triu(np.ones(corr_matrix.shape), k=1).astype(bool)  # Get the upper triangle of the matrix\n    upper_corr_matrix = corr_matrix.where(upper_triangle)  # Select only the upper triangle\n    \n    # Find index of features that are highly correlated\n    correlated_features = [column for column in upper_corr_matrix.columns if any(upper_corr_matrix[column] > threshold_corr)]\n    \n    # If target_col is specified, remove it from correlated features\n    if target_col and target_col in correlated_features:\n        correlated_features.remove(target_col)\n    \n    return correlated_features\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:53.951129Z","iopub.execute_input":"2024-09-28T17:51:53.951584Z","iopub.status.idle":"2024-09-28T17:51:53.962866Z","shell.execute_reply.started":"2024-09-28T17:51:53.951535Z","shell.execute_reply":"2024-09-28T17:51:53.960837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming `train` is your DataFrame, `threshold_corr` is your correlation threshold, and `target_col` is your target column\nthreshold_corr = 0.85  # Example threshold for high correlation\ntarget_col = 'Problematic Internet Uses'  # Example target column\n\ncorrelated_features = correlated_columns(train, threshold_corr, target_col) \ndrop_cols = []  # List of other columns you want to drop\ndropped_cols = np.unique(np.concatenate((drop_cols, correlated_features)))\n\nprint(\"Columns to drop due to high correlation:\", dropped_cols)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:53.964626Z","iopub.execute_input":"2024-09-28T17:51:53.965039Z","iopub.status.idle":"2024-09-28T17:51:54.015347Z","shell.execute_reply.started":"2024-09-28T17:51:53.965002Z","shell.execute_reply":"2024-09-28T17:51:54.014167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.016992Z","iopub.execute_input":"2024-09-28T17:51:54.017479Z","iopub.status.idle":"2024-09-28T17:51:54.024784Z","shell.execute_reply.started":"2024-09-28T17:51:54.017428Z","shell.execute_reply":"2024-09-28T17:51:54.023344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns = train.columns.str.strip()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.026303Z","iopub.execute_input":"2024-09-28T17:51:54.026793Z","iopub.status.idle":"2024-09-28T17:51:54.038232Z","shell.execute_reply.started":"2024-09-28T17:51:54.026723Z","shell.execute_reply":"2024-09-28T17:51:54.036778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_col = 'PreInt_EduHx-computerinternet_hoursday'\ny = train[target_col]\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.039933Z","iopub.execute_input":"2024-09-28T17:51:54.040932Z","iopub.status.idle":"2024-09-28T17:51:54.048182Z","shell.execute_reply.started":"2024-09-28T17:51:54.040885Z","shell.execute_reply":"2024-09-28T17:51:54.047089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update to the appropriate target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'\ny = train[target_col]\n\n# Check for correlated features with the new target column\nthreshold_corr = 0.85  # Example threshold for high correlation\ncorrelated_features = correlated_columns(train, threshold_corr, target_col) \ndrop_cols = []  # List of other columns you want to drop\ndropped_cols = np.unique(np.concatenate((drop_cols, correlated_features)))\n\nprint(\"Columns to drop due to high correlation:\", dropped_cols)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.049625Z","iopub.execute_input":"2024-09-28T17:51:54.050019Z","iopub.status.idle":"2024-09-28T17:51:54.099731Z","shell.execute_reply.started":"2024-09-28T17:51:54.049982Z","shell.execute_reply":"2024-09-28T17:51:54.098623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop(target_col, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.101231Z","iopub.execute_input":"2024-09-28T17:51:54.102437Z","iopub.status.idle":"2024-09-28T17:51:54.114180Z","shell.execute_reply.started":"2024-09-28T17:51:54.102383Z","shell.execute_reply":"2024-09-28T17:51:54.112633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef feature_correlation(X, y):\n    \"\"\"\n    Computes and visualizes the correlation between features and the target variable.\n    \n    Parameters:\n    X (pd.DataFrame): DataFrame containing the features.\n    y (pd.Series): Series containing the target variable.\n    \"\"\"\n    # Combine X and y into a single DataFrame for correlation computation\n    df_combined = pd.concat([X, y], axis=1)\n    \n    # Compute the correlation matrix\n    correlation_matrix = df_combined.corr()\n    \n    # Get correlations of features with the target variable\n    target_corr = correlation_matrix[y.name].sort_values(ascending=False)\n    \n    # Plotting the correlations\n    plt.figure(figsize=(20, 18))\n    sns.barplot(x=target_corr.index, y=target_corr.values, palette='viridis')\n    plt.title('Feature Correlation with Target Variable')\n    plt.xticks(rotation=90)\n    plt.ylabel('Correlation Coefficient')\n    plt.show()\n    \n    return target_corr\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.115734Z","iopub.execute_input":"2024-09-28T17:51:54.116130Z","iopub.status.idle":"2024-09-28T17:51:54.124656Z","shell.execute_reply.started":"2024-09-28T17:51:54.116091Z","shell.execute_reply":"2024-09-28T17:51:54.123163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your feature matrix X and target variable y\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\nX = train.drop(columns=[target_col])  # Assuming X contains all features except the target\n\n# Call the feature correlation function\ncorrelation_results = feature_correlation(X, y)\nprint(correlation_results)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:54.126403Z","iopub.execute_input":"2024-09-28T17:51:54.127406Z","iopub.status.idle":"2024-09-28T17:51:55.444233Z","shell.execute_reply.started":"2024-09-28T17:51:54.127353Z","shell.execute_reply":"2024-09-28T17:51:55.442962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.feature_selection import mutual_info_classif\n\ndef feature_correlation(X, y, method='mutual_info-classification'):\n    \"\"\"\n    Computes and visualizes feature correlation with the target variable \n    using the specified method.\n    \n    Parameters:\n    X (pd.DataFrame): DataFrame containing the features.\n    y (pd.Series): Series containing the target variable.\n    method (str): The method for calculating correlation. Currently supports\n                   'mutual_info-classification'.\n                   \n    Returns:\n    pd.Series: Series containing the correlation values for each feature.\n    \"\"\"\n    if method == 'mutual_info-classification':\n        # Calculate mutual information\n        mi = mutual_info_classif(X, y, discrete_features='auto')\n        mi_series = pd.Series(mi, index=X.columns).sort_values(ascending=False)\n        \n        # Plotting the mutual information scores\n        plt.figure(figsize=(20, 18))\n        sns.barplot(x=mi_series.index, y=mi_series.values, palette='viridis')\n        plt.title('Mutual Information Scores with Target Variable')\n        plt.xticks(rotation=90)\n        plt.ylabel('Mutual Information Score')\n        plt.show()\n        \n        return mi_series\n\n    else:\n        raise ValueError(\"Method not recognized. Please use 'mutual_info-classification'.\")\n\n# Define your feature matrix X and target variable y\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\nX = train.drop(columns=[target_col])  # Assuming X contains all features except the target\n\n# Call the feature correlation function with mutual information\nmi_results = feature_correlation(X, y, method='mutual_info-classification')\nprint(mi_results)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:55.445732Z","iopub.execute_input":"2024-09-28T17:51:55.446095Z","iopub.status.idle":"2024-09-28T17:51:57.494549Z","shell.execute_reply.started":"2024-09-28T17:51:55.446061Z","shell.execute_reply":"2024-09-28T17:51:57.493277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef plot_sns_corr_class(df, target_col):\n    \"\"\"\n    Plots a heatmap of the correlation matrix between features and the target variable.\n    \n    Parameters:\n    df (pd.DataFrame): DataFrame containing the features and target variable.\n    target_col (str): The name of the target variable column.\n    \"\"\"\n    # Calculate the correlation matrix\n    correlation_matrix = df.corr()\n    \n    # Extract the correlation of features with the target variable\n    target_corr = correlation_matrix[target_col].sort_values(ascending=False)\n    \n    # Create a heatmap for visualization\n    plt.figure(figsize=(20, 18))\n    sns.heatmap(target_corr.to_frame(), annot=True, cmap='coolwarm', linewidths=0.5)\n    plt.title(f'Correlation with Target Variable: {target_col}')\n    plt.ylabel('Features')\n    plt.show()\n\n    return target_corr\n\n# Define your target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update this to your target variable\n\n# Call the correlation plotting function\ncorrelation_results = plot_sns_corr_class(train, target_col)\nprint(correlation_results)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:57.495948Z","iopub.execute_input":"2024-09-28T17:51:57.496361Z","iopub.status.idle":"2024-09-28T17:51:58.711283Z","shell.execute_reply.started":"2024-09-28T17:51:57.496323Z","shell.execute_reply":"2024-09-28T17:51:58.710075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nsns.heatmap(train.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:58.713141Z","iopub.execute_input":"2024-09-28T17:51:58.713617Z","iopub.status.idle":"2024-09-28T17:51:59.630409Z","shell.execute_reply.started":"2024-09-28T17:51:58.713563Z","shell.execute_reply":"2024-09-28T17:51:59.629097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.tree import DecisionTreeClassifier\n\ndef feature_importances(model, X, y):\n    \"\"\"\n    Fit a decision tree model and plot feature importances.\n    \n    Parameters:\n    model: A scikit-learn classifier object (e.g., DecisionTreeClassifier).\n    X (pd.DataFrame): DataFrame containing the features.\n    y (pd.Series): Series containing the target variable.\n    \"\"\"\n    # Fit the model\n    model.fit(X, y)\n    \n    # Get feature importances\n    importances = model.feature_importances_\n    \n    # Create a DataFrame for visualization\n    feature_importance_df = pd.DataFrame({\n        'Feature': X.columns,\n        'Importance': importances\n    }).sort_values(by='Importance', ascending=False)\n    \n    # Plotting feature importances\n    plt.figure(figsize=(18, 16))\n    sns.barplot(x='Importance', y='Feature', data=feature_importance_df, palette='viridis')\n    plt.title('Feature Importances from Decision Tree Classifier')\n    plt.xlabel('Importance Score')\n    plt.ylabel('Features')\n    plt.show()\n    \n    return feature_importance_df\n\n# Define your target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\nX = train.drop(columns=[target_col])  # Assuming X contains all features except the target\n\n# Call the feature importances function\nimportance_results = feature_importances(DecisionTreeClassifier(), X, y)\nprint(importance_results)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:51:59.632027Z","iopub.execute_input":"2024-09-28T17:51:59.632480Z","iopub.status.idle":"2024-09-28T17:52:00.817393Z","shell.execute_reply.started":"2024-09-28T17:51:59.632433Z","shell.execute_reply":"2024-09-28T17:52:00.816267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LogisticRegression\n\ndef feature_importances_logistic(model, X, y):\n    \"\"\"\n    Fit a logistic regression model with elastic net penalty and plot feature importances.\n    \n    Parameters:\n    model: A scikit-learn LogisticRegression object.\n    X (pd.DataFrame): DataFrame containing the features.\n    y (pd.Series): Series containing the target variable.\n    \"\"\"\n    # Fit the model\n    model.fit(X, y)\n    \n    # Get feature importances (coefficients)\n    importances = model.coef_[0]  # Get the coefficients for the binary classification\n    feature_importance_df = pd.DataFrame({\n        'Feature': X.columns,\n        'Importance': importances\n    }).sort_values(by='Importance', ascending=False)\n    \n    # Plotting feature importances\n    plt.figure(figsize=(18, 16))\n    sns.barplot(x='Importance', y='Feature', data=feature_importance_df, palette='viridis')\n    plt.title('Feature Importances from Logistic Regression')\n    plt.xlabel('Importance Score')\n    plt.ylabel('Features')\n    plt.axvline(0, color='red', linestyle='--')  # Line at 0 for reference\n    plt.show()\n    \n    return feature_importance_df\n\nfrom sklearn.linear_model import LogisticRegression\n\n# Define your target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\nX = train.drop(columns=[target_col])  # Assuming X contains all features except the target\n\n# Call the feature importances function with logistic regression\nimportance_results = feature_importances_logistic(\n    LogisticRegression(penalty='elasticnet', solver='saga', l1_ratio=0.5, max_iter=10000), \n    X, y\n)\nprint(importance_results)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:00.818842Z","iopub.execute_input":"2024-09-28T17:52:00.819297Z","iopub.status.idle":"2024-09-28T17:52:51.882127Z","shell.execute_reply.started":"2024-09-28T17:52:00.819246Z","shell.execute_reply":"2024-09-28T17:52:51.880972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.decomposition import PCA\n\ndef pca_decomposition(X, y, projection=3):\n    \"\"\"\n    Perform PCA decomposition and plot the first two or three principal components.\n\n    Parameters:\n    X (pd.DataFrame): DataFrame containing the features.\n    y (pd.Series): Series containing the target variable (used for color coding).\n    projection (int): Number of principal components to plot (2 or 3).\n    \"\"\"\n    # Initialize PCA\n    pca = PCA(n_components=projection)\n    X_pca = pca.fit_transform(X)\n    \n    # Create a DataFrame for PCA results\n    pca_df = pd.DataFrame(data=X_pca, columns=[f'PC{i+1}' for i in range(projection)])\n    pca_df['Target'] = y.reset_index(drop=True)\n\n    # Plotting\n    if projection == 2:\n        plt.figure(figsize=(10, 8))\n        sns.scatterplot(x='PC1', y='PC2', hue='Target', data=pca_df, palette='viridis', alpha=0.7)\n        plt.title('PCA Result (2D)')\n        plt.xlabel('Principal Component 1')\n        plt.ylabel('Principal Component 2')\n        plt.legend(title='Target')\n        plt.grid(True)\n        plt.show()\n    \n    elif projection == 3:\n        fig = plt.figure(figsize=(10, 8))\n        ax = fig.add_subplot(111, projection='3d')\n        scatter = ax.scatter(pca_df['PC1'], pca_df['PC2'], pca_df['PC3'], c=pca_df['Target'], cmap='viridis', alpha=0.7)\n        plt.title('PCA Result (3D)')\n        ax.set_xlabel('Principal Component 1')\n        ax.set_ylabel('Principal Component 2')\n        ax.set_zlabel('Principal Component 3')\n        plt.colorbar(scatter, label='Target')\n        plt.show()\n    \n    else:\n        raise ValueError(\"Projection must be either 2 or 3.\")\n\n    return pca_df\n\n# Define your target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\nX = train.drop(columns=[target_col])  # Assuming X contains all features except the target\n\n# Call the PCA decomposition function\npca_results = pca_decomposition(X, y.astype(int), projection=3)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:51.883576Z","iopub.execute_input":"2024-09-28T17:52:51.883945Z","iopub.status.idle":"2024-09-28T17:52:52.422999Z","shell.execute_reply.started":"2024-09-28T17:52:51.883906Z","shell.execute_reply":"2024-09-28T17:52:52.421771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\ndef class_balance(y):\n    \"\"\"\n    Visualizes the balance of classes in the target variable.\n    \n    Parameters:\n    y (pd.Series): Series containing the target variable.\n    \"\"\"\n    plt.figure(figsize=(10, 6))\n    sns.countplot(x=y, palette='viridis')\n    plt.title('Class Balance')\n    plt.xlabel('Classes')\n    plt.ylabel('Count')\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.424678Z","iopub.execute_input":"2024-09-28T17:52:52.425213Z","iopub.status.idle":"2024-09-28T17:52:52.433986Z","shell.execute_reply.started":"2024-09-28T17:52:52.425158Z","shell.execute_reply":"2024-09-28T17:52:52.432827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef shannon_entropy(y):\n    \"\"\"\n    Calculates the Shannon entropy of the target variable.\n    \n    Parameters:\n    y (pd.Series): Series containing the target variable.\n    \n    Returns:\n    float: The Shannon entropy value.\n    \"\"\"\n    # Calculate class probabilities\n    probabilities = y.value_counts(normalize=True)\n    \n    # Calculate Shannon entropy\n    entropy = -np.sum(probabilities * np.log(probabilities + 1e-9))  # Adding small value to avoid log(0)\n    \n    return entropy\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.435597Z","iopub.execute_input":"2024-09-28T17:52:52.436021Z","iopub.status.idle":"2024-09-28T17:52:52.449920Z","shell.execute_reply.started":"2024-09-28T17:52:52.435955Z","shell.execute_reply":"2024-09-28T17:52:52.448561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your target column\ntarget_col = 'PreInt_EduHx-computerinternet_hoursday'  # Update as needed\ny = train[target_col]\n\n# Check class balance\nclass_balance(y)\n\n# Calculate and print Shannon entropy\nentropy_value = shannon_entropy(y)\nprint('Entropy = ', entropy_value)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.451329Z","iopub.execute_input":"2024-09-28T17:52:52.452087Z","iopub.status.idle":"2024-09-28T17:52:52.756169Z","shell.execute_reply.started":"2024-09-28T17:52:52.452044Z","shell.execute_reply":"2024-09-28T17:52:52.755064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train[target_col]\nX = train.drop(target_col, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.757620Z","iopub.execute_input":"2024-09-28T17:52:52.757990Z","iopub.status.idle":"2024-09-28T17:52:52.768167Z","shell.execute_reply.started":"2024-09-28T17:52:52.757952Z","shell.execute_reply":"2024-09-28T17:52:52.767047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_features = len(X.columns.tolist())\nnb_targets = len(y.unique())\nlayer_size = nb_features + nb_targets + 2","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.769714Z","iopub.execute_input":"2024-09-28T17:52:52.770206Z","iopub.status.idle":"2024-09-28T17:52:52.776729Z","shell.execute_reply.started":"2024-09-28T17:52:52.770144Z","shell.execute_reply":"2024-09-28T17:52:52.775596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\n\ndef custom_split(X, y, test_size=0.2, threshold_entropy=None, undersampling=False, undersampler=None, random_state=None):\n    \"\"\"\n    Custom function to split data into training and testing sets.\n    \n    Parameters:\n    X (pd.DataFrame): Feature set.\n    y (pd.Series): Target variable.\n    test_size (float): Proportion of the dataset to include in the test split.\n    threshold_entropy (float): Threshold for entropy-based filtering (if applicable).\n    undersampling (bool): Whether to perform undersampling.\n    undersampler: Undersampler instance (if applicable).\n    random_state (int): Random seed for reproducibility.\n    \n    Returns:\n    X_train, X_test, y_train, y_test: Split datasets.\n    \"\"\"\n    \n    # Optionally check for entropy and filter the data if needed\n    if threshold_entropy is not None:\n        # Implement your entropy filtering logic here\n        pass  # Placeholder for entropy logic\n    \n    # Perform undersampling if specified\n    if undersampling:\n        # Implement your undersampling logic here\n        # This is a simple example for undersampling the majority class\n        X, y = resample(X, y, replace=False, n_samples=len(y), random_state=random_state)  # Example implementation\n\n    # Split the dataset into training and testing sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=test_size, random_state=random_state, stratify=y)\n    \n    return X_train, X_test, y_train, y_test\n# Assuming X and y are defined DataFrames/Series\ntest_size = 0.2\nthreshold_entropy = None  # Set your threshold if needed\nundersampling = False  # Set to True if you want to perform undersampling\nundersampler = None  # Specify your undersampler instance if you have one\nrandom_state = 42  # For reproducibility\n\n# Split the data\nX_train, X_test, y_train, y_test = custom_split(X, y, test_size=test_size, \n                                                  threshold_entropy=threshold_entropy, \n                                                  undersampling=undersampling, \n                                                  undersampler=undersampler, \n                                                  random_state=random_state)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.778141Z","iopub.execute_input":"2024-09-28T17:52:52.778605Z","iopub.status.idle":"2024-09-28T17:52:52.799854Z","shell.execute_reply.started":"2024-09-28T17:52:52.778556Z","shell.execute_reply":"2024-09-28T17:52:52.798363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_encoder = LabelEncoder() \ny_train = pd.Series(target_encoder.fit_transform(y_train)) \ny_test = pd.Series(target_encoder.transform(y_test))","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.801734Z","iopub.execute_input":"2024-09-28T17:52:52.802244Z","iopub.status.idle":"2024-09-28T17:52:52.810188Z","shell.execute_reply.started":"2024-09-28T17:52:52.802193Z","shell.execute_reply":"2024-09-28T17:52:52.809142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n\n# Define your parameters\ninput_shape = X_train.shape[1]  # Number of features\nlayer_size = 128  # Adjust as needed\nnb_targets = len(np.unique(y_train))  # Number of unique classes\n\n# Define your model architecture\ndef K_Class():\n    tf.keras.backend.clear_session()\n    model = Sequential()\n    model.add(BatchNormalization(input_shape=(input_shape,)))\n    \n    # First hidden layer\n    model.add(Dense(layer_size, activation='selu'))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    # Second hidden layer\n    model.add(Dense(layer_size, activation='selu'))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    # Third hidden layer (optional)\n    model.add(Dense(layer_size, activation='selu'))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    # Output layer\n    model.add(Dense(nb_targets, activation='softmax'))\n\n    # Compile the model\n    model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model\n# Create and fit the model\ndef fit_model(X_train, y_train):\n    model = K_Class()\n    es = EarlyStopping(monitor='val_loss', mode='auto', verbose=1, patience=20)\n    \n    # Convert y_train to categorical\n    y_train_categorical = to_categorical(y_train)\n    \n    # Fit the model\n    model.fit(X_train, y_train_categorical, batch_size=64, epochs=2000, validation_split=0.1, callbacks=[es], verbose=1)\n    return model\n\n# Fit the model to your training data\nmodel = fit_model(X_train, y_train)\n\n# After fitting, you can evaluate or make predictions\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:52:52.811455Z","iopub.execute_input":"2024-09-28T17:52:52.811809Z","iopub.status.idle":"2024-09-28T17:53:04.128189Z","shell.execute_reply.started":"2024-09-28T17:52:52.811774Z","shell.execute_reply":"2024-09-28T17:53:04.126825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries for Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\n\n# Define and fit the Random Forest model\ndef fit_random_forest(X_train, y_train, n_estimators=1000, random_state=42):\n    # Initialize the model\n    rf_model = RandomForestClassifier(n_estimators=n_estimators, random_state=random_state)\n    \n    # Fit the model\n    rf_model.fit(X_train, y_train)\n    \n    return rf_model\n\n# Fit the Random Forest model to your training data\nrf_model = fit_random_forest(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = rf_model.predict(X_test)\n\n# Evaluate the model\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred))\nprint(\"Accuracy Score:\", accuracy_score(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:53:04.130346Z","iopub.execute_input":"2024-09-28T17:53:04.130725Z","iopub.status.idle":"2024-09-28T17:53:13.307234Z","shell.execute_reply.started":"2024-09-28T17:53:04.130687Z","shell.execute_reply":"2024-09-28T17:53:13.305961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries for SVM\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Define and fit the SVM model\ndef fit_svm(X_train, y_train, kernel='rbf', C=1.0, random_state=42):\n    # Create a pipeline with standard scaling and SVM\n    svm_pipeline = make_pipeline(StandardScaler(), SVC(kernel=kernel, C=C, random_state=random_state))\n    \n    # Fit the model\n    svm_pipeline.fit(X_train, y_train)\n    \n    return svm_pipeline\n\n# Fit the SVM model to your training data\nsvm_model = fit_svm(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = svm_model.predict(X_test)\n\n# Evaluate the model\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred))\nprint(\"Accuracy Score:\", accuracy_score(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T17:53:13.308624Z","iopub.execute_input":"2024-09-28T17:53:13.309288Z","iopub.status.idle":"2024-09-28T17:53:13.798027Z","shell.execute_reply.started":"2024-09-28T17:53:13.309248Z","shell.execute_reply":"2024-09-28T17:53:13.796720Z"},"trusted":true},"execution_count":null,"outputs":[]}]}