{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-04T10:16:59.435630Z","iopub.execute_input":"2024-10-04T10:16:59.437070Z","iopub.status.idle":"2024-10-04T10:17:00.050621Z","shell.execute_reply.started":"2024-10-04T10:16:59.436997Z","shell.execute_reply":"2024-10-04T10:17:00.049249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('sss')","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.052968Z","iopub.execute_input":"2024-10-04T10:17:00.053419Z","iopub.status.idle":"2024-10-04T10:17:00.060856Z","shell.execute_reply.started":"2024-10-04T10:17:00.053374Z","shell.execute_reply":"2024-10-04T10:17:00.059160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## check out the files","metadata":{}},{"cell_type":"code","source":"df_train=pd.read_csv(r'/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test =pd.read_csv(r'/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_sub=pd.read_csv(r'/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ndata_dict= pd.read_csv(r'/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.063081Z","iopub.execute_input":"2024-10-04T10:17:00.063749Z","iopub.status.idle":"2024-10-04T10:17:00.148842Z","shell.execute_reply.started":"2024-10-04T10:17:00.063671Z","shell.execute_reply":"2024-10-04T10:17:00.147066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.152002Z","iopub.execute_input":"2024-10-04T10:17:00.152440Z","iopub.status.idle":"2024-10-04T10:17:00.186831Z","shell.execute_reply.started":"2024-10-04T10:17:00.152389Z","shell.execute_reply":"2024-10-04T10:17:00.185209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.188626Z","iopub.execute_input":"2024-10-04T10:17:00.189096Z","iopub.status.idle":"2024-10-04T10:17:00.222404Z","shell.execute_reply.started":"2024-10-04T10:17:00.189052Z","shell.execute_reply":"2024-10-04T10:17:00.221144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.224231Z","iopub.execute_input":"2024-10-04T10:17:00.224691Z","iopub.status.idle":"2024-10-04T10:17:00.238292Z","shell.execute_reply.started":"2024-10-04T10:17:00.224637Z","shell.execute_reply":"2024-10-04T10:17:00.236597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.240351Z","iopub.execute_input":"2024-10-04T10:17:00.240895Z","iopub.status.idle":"2024-10-04T10:17:00.270450Z","shell.execute_reply.started":"2024-10-04T10:17:00.240839Z","shell.execute_reply":"2024-10-04T10:17:00.268911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.272150Z","iopub.execute_input":"2024-10-04T10:17:00.272600Z","iopub.status.idle":"2024-10-04T10:17:00.291825Z","shell.execute_reply.started":"2024-10-04T10:17:00.272557Z","shell.execute_reply":"2024-10-04T10:17:00.290152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## check for null values","metadata":{}},{"cell_type":"code","source":"null_values = df_train.isnull().sum().sort_values(ascending=False)\nprint(null_values.to_string())","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.293927Z","iopub.execute_input":"2024-10-04T10:17:00.294517Z","iopub.status.idle":"2024-10-04T10:17:00.313395Z","shell.execute_reply.started":"2024-10-04T10:17:00.294452Z","shell.execute_reply":"2024-10-04T10:17:00.311814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_raw=df_train.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.319263Z","iopub.execute_input":"2024-10-04T10:17:00.320138Z","iopub.status.idle":"2024-10-04T10:17:00.328443Z","shell.execute_reply.started":"2024-10-04T10:17:00.320086Z","shell.execute_reply":"2024-10-04T10:17:00.326967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('length of df', len(df_train_raw))","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.330077Z","iopub.execute_input":"2024-10-04T10:17:00.330498Z","iopub.status.idle":"2024-10-04T10:17:00.343560Z","shell.execute_reply.started":"2024-10-04T10:17:00.330447Z","shell.execute_reply":"2024-10-04T10:17:00.341899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's remove all the columns where the predicatble value is null","metadata":{}},{"cell_type":"code","source":"df_train.dropna(subset=['sii'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.345463Z","iopub.execute_input":"2024-10-04T10:17:00.346683Z","iopub.status.idle":"2024-10-04T10:17:00.359988Z","shell.execute_reply.started":"2024-10-04T10:17:00.346621Z","shell.execute_reply":"2024-10-04T10:17:00.358389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('length of df', len(df_train))","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.362292Z","iopub.execute_input":"2024-10-04T10:17:00.362904Z","iopub.status.idle":"2024-10-04T10:17:00.372974Z","shell.execute_reply.started":"2024-10-04T10:17:00.362847Z","shell.execute_reply":"2024-10-04T10:17:00.371434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's do encoding for all the categorical columns, let's first check the number of null values for each of the columns","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.374804Z","iopub.execute_input":"2024-10-04T10:17:00.376161Z","iopub.status.idle":"2024-10-04T10:17:00.386162Z","shell.execute_reply.started":"2024-10-04T10:17:00.376048Z","shell.execute_reply":"2024-10-04T10:17:00.384815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = df_train.select_dtypes(include=['object']).columns\nother_cols = df_train.select_dtypes(exclude=['object']).columns","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.387826Z","iopub.execute_input":"2024-10-04T10:17:00.388169Z","iopub.status.idle":"2024-10-04T10:17:00.400937Z","shell.execute_reply.started":"2024-10-04T10:17:00.388132Z","shell.execute_reply":"2024-10-04T10:17:00.399489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find the null values in the categorical columns","metadata":{}},{"cell_type":"code","source":"df_train[categorical_cols].isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.402680Z","iopub.execute_input":"2024-10-04T10:17:00.403234Z","iopub.status.idle":"2024-10-04T10:17:00.422242Z","shell.execute_reply.started":"2024-10-04T10:17:00.403188Z","shell.execute_reply":"2024-10-04T10:17:00.420524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# for a start let's remove the  columns that have high null values(>1200)","metadata":{}},{"cell_type":"code","source":"df_train.drop([\n    'PAQ_A-Season',\n    'Fitness_Endurance-Season',\n    'PAQ_C-Season'\n], inplace=True,axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.424192Z","iopub.execute_input":"2024-10-04T10:17:00.424695Z","iopub.status.idle":"2024-10-04T10:17:00.435674Z","shell.execute_reply.started":"2024-10-04T10:17:00.424649Z","shell.execute_reply":"2024-10-04T10:17:00.434197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# for the null values between 17 and 1200 let's replace the null values with unknown and then perform label encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# List of columns you want to apply label encoding to (removing duplicates)\ncategorical_cols_two = ['Physical-Season', 'FGC-Season', 'PreInt_EduHx-Season', \n                        'Basic_Demos-Enroll_Season', 'PCIAT-Season', 'SDS-Season', 'id','BIA-Season','CGAS-Season']\n\n# Fill null values with 'Unknown' for each column\ndf_train[categorical_cols_two] = df_train[categorical_cols_two].fillna('Unknown')\n\n# Initialize LabelEncoder\nencoder = LabelEncoder()\n\n# Apply Label Encoding to each column individually\nfor col in categorical_cols_two:\n    # Check if the column exists in the DataFrame\n    if col in df_train.columns:\n        df_train[col] = encoder.fit_transform(df_train[col])\n\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.437619Z","iopub.execute_input":"2024-10-04T10:17:00.439033Z","iopub.status.idle":"2024-10-04T10:17:00.516023Z","shell.execute_reply.started":"2024-10-04T10:17:00.438971Z","shell.execute_reply":"2024-10-04T10:17:00.514599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's do feature engineering for the non-categorical values","metadata":{}},{"cell_type":"markdown","source":"# find the null values in the non-categorical valuea","metadata":{}},{"cell_type":"code","source":"# Print the null value counts for specific columns in other_cols without truncating\nprint(df_train[other_cols].isnull().sum().sort_values(ascending=False).to_string())","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.517980Z","iopub.execute_input":"2024-10-04T10:17:00.518402Z","iopub.status.idle":"2024-10-04T10:17:00.531307Z","shell.execute_reply.started":"2024-10-04T10:17:00.518360Z","shell.execute_reply":"2024-10-04T10:17:00.529791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's drop all the columns whose null values are more than 1200","metadata":{}},{"cell_type":"code","source":"df_train.drop([\n    'PAQ_A-PAQ_A_Total',\n'Physical-Waist_Circumference',\n'Fitness_Endurance-Time_Sec',\n'Fitness_Endurance-Time_Mins',\n'Fitness_Endurance-Max_Stage',\n'FGC-FGC_GSD_Zone',\n'FGC-FGC_GSND_Zone',\n'FGC-FGC_GSD',\n'FGC-FGC_GSND',\n'PAQ_C-PAQ_C_Total'\n],inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.533273Z","iopub.execute_input":"2024-10-04T10:17:00.533792Z","iopub.status.idle":"2024-10-04T10:17:00.544161Z","shell.execute_reply.started":"2024-10-04T10:17:00.533734Z","shell.execute_reply":"2024-10-04T10:17:00.542813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# For the ones between 2 and 1200 let's fill the null values with mean of the field value","metadata":{}},{"cell_type":"code","source":"# List of numeric columns to fill with mean\nnumeric_columns = [\n'Physical-Diastolic_BP',\n'Physical-Systolic_BP',\n'Physical-HeartRate',\n'SDS-SDS_Total_T',\n'Physical-BMI',\n'SDS-SDS_Total_Raw',\n'Physical-Height',\n'Physical-Weight',\n'PreInt_EduHx-computerinternet_hoursday',\n'PCIAT-PCIAT_17',\n'PCIAT-PCIAT_18',\n'PCIAT-PCIAT_16',\n'PCIAT-PCIAT_13',\n'PCIAT-PCIAT_05',\n'PCIAT-PCIAT_07',\n'PCIAT-PCIAT_19',\n'PCIAT-PCIAT_15',\n'PCIAT-PCIAT_09',\n'PCIAT-PCIAT_08',\n'PCIAT-PCIAT_12',\n'PCIAT-PCIAT_04',\n'PCIAT-PCIAT_03',\n'PCIAT-PCIAT_14',\n'PCIAT-PCIAT_06',\n'PCIAT-PCIAT_10',\n'PCIAT-PCIAT_01',\n'PCIAT-PCIAT_02',\n'PCIAT-PCIAT_11',\n'PCIAT-PCIAT_20',\n 'BIA-BIA_BMI',\n'BIA-BIA_BMR',\n'BIA-BIA_DEE',\n'BIA-BIA_ECW',\n'BIA-BIA_FFMI',\n'BIA-BIA_FMI',\n'BIA-BIA_Fat',\n'BIA-BIA_Frame_num',\n'BIA-BIA_Activity_Level_num',\n'BIA-BIA_ICW',\n'BIA-BIA_LDM',\n'BIA-BIA_LST',\n'BIA-BIA_SMM',\n'BIA-BIA_TBW',\n'BIA-BIA_BMC',\n'BIA-BIA_FFM',\n'FGC-FGC_PU_Zone',\n'FGC-FGC_SRL_Zone',\n'FGC-FGC_SRR_Zone',\n'FGC-FGC_CU_Zone',\n'FGC-FGC_TL_Zone',\n'FGC-FGC_PU',\n'FGC-FGC_SRL',\n'FGC-FGC_SRR',\n'FGC-FGC_TL',\n'FGC-FGC_CU',\n'CGAS-CGAS_Score'   \n\n]  # Replace with your actual column names\n\n# Fill null values with the mean for the specified numeric columns\ndf_train[numeric_columns] = df_train[numeric_columns].fillna(df_train[numeric_columns].mean())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.545925Z","iopub.execute_input":"2024-10-04T10:17:00.546434Z","iopub.status.idle":"2024-10-04T10:17:00.593446Z","shell.execute_reply.started":"2024-10-04T10:17:00.546377Z","shell.execute_reply":"2024-10-04T10:17:00.592245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's define a correlation matrix with the predicted column sii","metadata":{}},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = df_train.corr()\n\n# Get the correlation of the 'sii' column with all other columns\ncorrelation_with_sii = correlation_matrix['sii']\n\ncorrelation_with_sii.sort_values(ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.595215Z","iopub.execute_input":"2024-10-04T10:17:00.595667Z","iopub.status.idle":"2024-10-04T10:17:00.647035Z","shell.execute_reply.started":"2024-10-04T10:17:00.595624Z","shell.execute_reply":"2024-10-04T10:17:00.645501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# let's filter the low correlation columns filter all the fields are that greater than < 0.5","metadata":{}},{"cell_type":"code","source":"# Identify columns to drop (correlation < 0.1)\ncolumns_to_drop = correlation_with_sii[correlation_with_sii < 0].index.tolist()\n\n# Drop the identified columns from df_train\ndf_train.drop(columns=columns_to_drop, inplace=True, errors='ignore')","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.648934Z","iopub.execute_input":"2024-10-04T10:17:00.649407Z","iopub.status.idle":"2024-10-04T10:17:00.662543Z","shell.execute_reply.started":"2024-10-04T10:17:00.649361Z","shell.execute_reply":"2024-10-04T10:17:00.661212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's further drop the redudant columns(if two columns have similar correlation we will pick one)","metadata":{}},{"cell_type":"code","source":"df_train.drop([\n    'PCIAT-PCIAT_14',\n    'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_01',\n    'PCIAT-PCIAT_11'],inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.665055Z","iopub.execute_input":"2024-10-04T10:17:00.665630Z","iopub.status.idle":"2024-10-04T10:17:00.677753Z","shell.execute_reply.started":"2024-10-04T10:17:00.665571Z","shell.execute_reply":"2024-10-04T10:17:00.676263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Perform train test split with the train data and train it with some basic models","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Step 1: Import necessary libraries\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\n\n# Assuming df_train is your DataFrame and 'sii' is the target column\n# Step 2: Split the data into features and target variable\nX = df_train.drop(columns=['sii'])  # Features\ny = df_train['sii']                 # Target variable\n\n# Step 3: Train-Test Split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Optional: Scale the features (important for some models)\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Step 4: Build and train models\nmodels = {\n    'Logistic Regression': LogisticRegression(),\n    'Random Forest': RandomForestClassifier(),\n    'Support Vector Classifier': SVC(),\n    'Decision Tree': DecisionTreeClassifier(),\n    'K-Nearest Neighbors': KNeighborsClassifier()\n}\n\n# Step 5: Train the models, calculate accuracy, and track the best model\nbest_model = None\nbest_model_name = None\nbest_accuracy = 0.0\nbest_report = \"\"\n\nfor model_name, model in models.items():\n    # Fit the model\n    model.fit(X_train, y_train)\n    \n    # Predict on the test set\n    y_pred = model.predict(X_test)\n    \n    # Calculate accuracy (you can choose another metric if needed)\n    accuracy = accuracy_score(y_test, y_pred)\n    \n    # Print the classification report for the current model\n    print(f\"Classification Report for {model_name}:\\n\")\n    report = classification_report(y_test, y_pred)\n    print(report)\n    print(\"=\" * 80)\n    \n    # Check if this is the best model so far\n    if accuracy > best_accuracy:\n        best_model = model\n        best_model_name = model_name\n        best_accuracy = accuracy\n        best_report = report\n\n# Step 6: Print the best model and its classification report\nprint(f\"Best Model: {best_model_name} with accuracy: {best_accuracy:.4f}\\n\")\nprint(\"Classification Report for the Best Model:\\n\")\nprint(best_report)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:00.679541Z","iopub.execute_input":"2024-10-04T10:17:00.679936Z","iopub.status.idle":"2024-10-04T10:17:01.810954Z","shell.execute_reply.started":"2024-10-04T10:17:00.679895Z","shell.execute_reply":"2024-10-04T10:17:01.809801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:01.812278Z","iopub.execute_input":"2024-10-04T10:17:01.812660Z","iopub.status.idle":"2024-10-04T10:17:01.838971Z","shell.execute_reply.started":"2024-10-04T10:17:01.812619Z","shell.execute_reply":"2024-10-04T10:17:01.837787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:01.840371Z","iopub.execute_input":"2024-10-04T10:17:01.840746Z","iopub.status.idle":"2024-10-04T10:17:02.049697Z","shell.execute_reply.started":"2024-10-04T10:17:01.840676Z","shell.execute_reply":"2024-10-04T10:17:02.048059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's perform the same cleaning in the test file and attempt attempt prediction with 100% train data","metadata":{}},{"cell_type":"code","source":"original_ids = df_test['id'].copy()  # copy the id as it would be useful","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.057270Z","iopub.execute_input":"2024-10-04T10:17:02.057796Z","iopub.status.idle":"2024-10-04T10:17:02.064965Z","shell.execute_reply.started":"2024-10-04T10:17:02.057747Z","shell.execute_reply":"2024-10-04T10:17:02.063272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['id']","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.066632Z","iopub.execute_input":"2024-10-04T10:17:02.067154Z","iopub.status.idle":"2024-10-04T10:17:02.082784Z","shell.execute_reply.started":"2024-10-04T10:17:02.067103Z","shell.execute_reply":"2024-10-04T10:17:02.081084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.084669Z","iopub.execute_input":"2024-10-04T10:17:02.086121Z","iopub.status.idle":"2024-10-04T10:17:02.101599Z","shell.execute_reply.started":"2024-10-04T10:17:02.085982Z","shell.execute_reply":"2024-10-04T10:17:02.100151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## overall sample size of the test data is 14, i'll drop all the columns with null values more than 6","metadata":{}},{"cell_type":"code","source":"# Check for null values and their counts\nnull_counts = df_test.isnull().sum()\n\n# Identify columns with more than 5 null values\ncolumns_to_drop = null_counts[null_counts > 14].index\n\n# Drop these columns from df_test\ndf_test.drop(columns=columns_to_drop,inplace=True)\n\n# Optional: Display the cleaned DataFrame and confirm the drop\nprint(\"Dropped columns:\", columns_to_drop.tolist())\nprint(\"*\")\nprint(\"Remaining columns:\", df_test.columns.tolist())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.103508Z","iopub.execute_input":"2024-10-04T10:17:02.104068Z","iopub.status.idle":"2024-10-04T10:17:02.118767Z","shell.execute_reply.started":"2024-10-04T10:17:02.104021Z","shell.execute_reply":"2024-10-04T10:17:02.117062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's handle all the categorical columns with label encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Create a LabelEncoder instance\nlabel_encoder = LabelEncoder()\n\n# Select categorical columns in df_test\ncategorical_cols = df_test.select_dtypes(include=['object']).columns\n\n# Apply label encoding to each categorical column\nfor col in categorical_cols:\n    df_test[col] = label_encoder.fit_transform(df_test[col].astype(str))\n\n# Display the first few rows of the encoded DataFrame\ndf_test.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.120919Z","iopub.execute_input":"2024-10-04T10:17:02.121355Z","iopub.status.idle":"2024-10-04T10:17:02.164130Z","shell.execute_reply.started":"2024-10-04T10:17:02.121315Z","shell.execute_reply":"2024-10-04T10:17:02.162529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## lets handle the non-categorical columns by using average imputation to handle the null values","metadata":{}},{"cell_type":"code","source":"non_categorical_cols = df_test.select_dtypes(exclude=['object']).columns\n\n# Perform mean imputation for each non-categorical column\nfor col in non_categorical_cols:\n    mean_value = df_test[col].mean()  # Calculate the mean of the column\n    df_test[col].fillna(mean_value, inplace=True)  # Fill nulls with the mean value","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.166467Z","iopub.execute_input":"2024-10-04T10:17:02.167227Z","iopub.status.idle":"2024-10-04T10:17:02.195820Z","shell.execute_reply.started":"2024-10-04T10:17:02.167172Z","shell.execute_reply":"2024-10-04T10:17:02.193803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_train['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.197505Z","iopub.execute_input":"2024-10-04T10:17:02.197974Z","iopub.status.idle":"2024-10-04T10:17:02.211515Z","shell.execute_reply.started":"2024-10-04T10:17:02.197927Z","shell.execute_reply":"2024-10-04T10:17:02.209883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_columns = df_train.drop(columns=['sii']).columns.intersection(df_test.columns)\nlen(common_columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.213458Z","iopub.execute_input":"2024-10-04T10:17:02.214555Z","iopub.status.idle":"2024-10-04T10:17:02.230448Z","shell.execute_reply.started":"2024-10-04T10:17:02.214477Z","shell.execute_reply":"2024-10-04T10:17:02.228980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_columns","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.232586Z","iopub.execute_input":"2024-10-04T10:17:02.233159Z","iopub.status.idle":"2024-10-04T10:17:02.242823Z","shell.execute_reply.started":"2024-10-04T10:17:02.233103Z","shell.execute_reply":"2024-10-04T10:17:02.241420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Align both datasets by selecting only the common columns\ncommon_columns = df_train.drop(columns=['sii']).columns.intersection(df_test.columns)\n\n# Now X_train_full and X_test_aligned have the same columns (features)\nX_train_full = df_train[common_columns]  # Features from training set\ny_train_full = df_train['sii']           # Target from training set\n\nX_test = df_test[common_columns]         # Features from test set\n\n# Optional: Scale the features (important for some models)\nscaler = StandardScaler()\nX_train_full_stand = scaler.fit_transform(X_train_full)\nX_test_full_stand = scaler.transform(X_test)\n\n# Step 4: Build and train models\nmodels = {\n    'Logistic Regression': LogisticRegression(),\n    'Random Forest': RandomForestClassifier(),\n    'Support Vector Classifier': SVC(),\n    'Decision Tree': DecisionTreeClassifier(),\n    'K-Nearest Neighbors': KNeighborsClassifier()\n}\n\n# Initialize variables to track the best model\nbest_model = None\nbest_model_name = None\nbest_accuracy = 0.0\nbest_y_pred = None\n\n# Step 5: Train the models, calculate accuracy, and select the best model\nfor model_name, model in models.items():\n    # Fit the model on the training data\n    model.fit(X_train_full_stand, y_train_full)\n    \n    # Predict on the test set\n    y_pred = model.predict(X_test_full_stand)\n    \n    # If you had ground truth labels for the test set, you could calculate the accuracy here.\n    # However, since you are testing on df_test without true labels, assume we focus on model selection.\n    # For demonstration, you could select based on training data performance (e.g., accuracy).\n    \n    # Evaluate model using accuracy on the training data (since test labels aren't available)\n    accuracy = model.score(X_train_full_stand, y_train_full)\n    \n    # Check if this is the best model so far\n    if accuracy > best_accuracy:\n        best_accuracy = accuracy\n        best_model = model\n        best_model_name = model_name\n        best_y_pred = y_pred\n\n# Step 6: Print the best model and its predictions\nprint(f\"Best Model: {best_model_name} with accuracy on training data: {best_accuracy:.4f}\")\nprint(\"\\nPredictions of the best model on the test set:\")\nprint(best_y_pred)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:02.244568Z","iopub.execute_input":"2024-10-04T10:17:02.245007Z","iopub.status.idle":"2024-10-04T10:17:04.782067Z","shell.execute_reply.started":"2024-10-04T10:17:02.244963Z","shell.execute_reply":"2024-10-04T10:17:04.780432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame with IDs and predictions\nresults_df = pd.DataFrame({\n    'id': original_ids,\n    'sii': best_y_pred\n})","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:04.783947Z","iopub.execute_input":"2024-10-04T10:17:04.784359Z","iopub.status.idle":"2024-10-04T10:17:04.791747Z","shell.execute_reply.started":"2024-10-04T10:17:04.784318Z","shell.execute_reply":"2024-10-04T10:17:04.790122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:04.793976Z","iopub.execute_input":"2024-10-04T10:17:04.795335Z","iopub.status.idle":"2024-10-04T10:17:04.814514Z","shell.execute_reply.started":"2024-10-04T10:17:04.795282Z","shell.execute_reply":"2024-10-04T10:17:04.812773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T10:17:04.816310Z","iopub.execute_input":"2024-10-04T10:17:04.816767Z","iopub.status.idle":"2024-10-04T10:17:04.827186Z","shell.execute_reply.started":"2024-10-04T10:17:04.816695Z","shell.execute_reply":"2024-10-04T10:17:04.825703Z"},"trusted":true},"execution_count":null,"outputs":[]}]}