{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n#For correlations\nfrom scipy.stats import chi2_contingency\nimport scipy.stats as stats\nfrom scipy import stats\nfrom statsmodels.formula.api import ols\nimport statsmodels.api as sm\nfrom sklearn import preprocessing\n\n#For entropy\nfrom scipy.special import entr\nfrom scipy.spatial.distance import cdist\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nfrom scipy.spatial.distance import pdist\nfrom scipy.spatial.distance import jaccard, squareform\n\nfrom itertools import product\n\nimport warnings\n\nimport matplotlib.pyplot as plt #For all plots\nimport seaborn as sn #heatmap plotting\n\n#for scaling\nfrom sklearn.preprocessing import StandardScaler\n\n#For linear regression\nfrom sklearn.linear_model import RidgeCV, Ridge\n\n#For KNN\nfrom sklearn.neighbors import KNeighborsClassifier, KNeighborsRegressor\n\n#for tests and accuracy\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import accuracy_score\nimport math","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:20.283127Z","iopub.execute_input":"2024-12-03T03:12:20.283794Z","iopub.status.idle":"2024-12-03T03:12:23.473981Z","shell.execute_reply.started":"2024-12-03T03:12:20.283756Z","shell.execute_reply":"2024-12-03T03:12:23.472863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TABLE INFO","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.476213Z","iopub.execute_input":"2024-12-03T03:12:23.476831Z","iopub.status.idle":"2024-12-03T03:12:23.567987Z","shell.execute_reply.started":"2024-12-03T03:12:23.476783Z","shell.execute_reply":"2024-12-03T03:12:23.566908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Replace the \"-\" with an empty space\nnewName = []\nfor column in train.columns:\n    columnName = column.replace(\"-\", \"\")\n    newName.append(columnName)\ntrain.columns = newName\n\nfor i, item in enumerate(dictionary[\"Field\"]):\n    item = item.replace(\"-\", \"\")\n    dictionary.loc[i, \"Field\"] = item","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.569280Z","iopub.execute_input":"2024-12-03T03:12:23.569601Z","iopub.status.idle":"2024-12-03T03:12:23.585268Z","shell.execute_reply.started":"2024-12-03T03:12:23.569569Z","shell.execute_reply":"2024-12-03T03:12:23.583994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_before = train.copy()\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.588261Z","iopub.execute_input":"2024-12-03T03:12:23.588765Z","iopub.status.idle":"2024-12-03T03:12:23.640159Z","shell.execute_reply.started":"2024-12-03T03:12:23.588713Z","shell.execute_reply":"2024-12-03T03:12:23.639139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.641519Z","iopub.execute_input":"2024-12-03T03:12:23.641975Z","iopub.status.idle":"2024-12-03T03:12:23.901525Z","shell.execute_reply.started":"2024-12-03T03:12:23.641922Z","shell.execute_reply":"2024-12-03T03:12:23.900340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_before = test.copy()\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.902786Z","iopub.execute_input":"2024-12-03T03:12:23.903193Z","iopub.status.idle":"2024-12-03T03:12:23.937584Z","shell.execute_reply.started":"2024-12-03T03:12:23.903160Z","shell.execute_reply":"2024-12-03T03:12:23.936434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:23.938897Z","iopub.execute_input":"2024-12-03T03:12:23.939264Z","iopub.status.idle":"2024-12-03T03:12:24.033353Z","shell.execute_reply.started":"2024-12-03T03:12:23.939231Z","shell.execute_reply":"2024-12-03T03:12:24.032288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dictionary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:24.035075Z","iopub.execute_input":"2024-12-03T03:12:24.035518Z","iopub.status.idle":"2024-12-03T03:12:24.050229Z","shell.execute_reply.started":"2024-12-03T03:12:24.035469Z","shell.execute_reply":"2024-12-03T03:12:24.048939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# UNIVERSAL FUNCTIONS","metadata":{}},{"cell_type":"code","source":"def get_data_type(target_column):\n    \n    if target_column == \"sii\":\n        return \"Categorical\"\n    \n    data_type = dictionary[dictionary['Field'] == target_column]['Type'].iloc[0]\n    if data_type == \"categorical int\" or data_type == \"str\":\n        return \"Categorical\"\n    else:\n        return \"Continuous\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:24.051951Z","iopub.execute_input":"2024-12-03T03:12:24.052294Z","iopub.status.idle":"2024-12-03T03:12:24.063072Z","shell.execute_reply.started":"2024-12-03T03:12:24.052262Z","shell.execute_reply":"2024-12-03T03:12:24.061762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def one_hot(table):\n    for column in table.columns:\n        if table[column].dtype == \"object\":\n            values = table[column].value_counts()\n            for i in range (len(values)):\n                table.replace(values.index[i], i+1, inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:24.067393Z","iopub.execute_input":"2024-12-03T03:12:24.067729Z","iopub.status.idle":"2024-12-03T03:12:24.082235Z","shell.execute_reply.started":"2024-12-03T03:12:24.067695Z","shell.execute_reply":"2024-12-03T03:12:24.080869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"one_hot(train)\none_hot(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:24.083435Z","iopub.execute_input":"2024-12-03T03:12:24.083812Z","iopub.status.idle":"2024-12-03T03:12:39.054210Z","shell.execute_reply.started":"2024-12-03T03:12:24.083778Z","shell.execute_reply":"2024-12-03T03:12:39.053148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GRAPHS","metadata":{}},{"cell_type":"code","source":"# Display counts of filled and empty values for each column for defined table\ndef filledIndividualPlot(table_name):\n    filled_counts = table_name.notna().sum()\n    empty_counts = table_name.isna().sum()\n\n    # Plotting\n    fig, ax = plt.subplots(figsize=(10, 15))\n\n    sorted_indices = filled_counts.sort_values(ascending=False).index\n    filled_counts = filled_counts[sorted_indices]\n    empty_counts = empty_counts[sorted_indices]\n\n    # Create stacked horizontal bars\n    ax.barh(sorted_indices, filled_counts, color='green', label='Filled')\n    ax.barh(sorted_indices, empty_counts, left=filled_counts, color='red', label='Empty')\n\n    # Labels and legend\n    ax.set_xlabel('Count')\n    ax.set_title('Count of Filled and Empty Values by Column')\n    ax.legend()\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.055623Z","iopub.execute_input":"2024-12-03T03:12:39.056068Z","iopub.status.idle":"2024-12-03T03:12:39.063526Z","shell.execute_reply.started":"2024-12-03T03:12:39.056022Z","shell.execute_reply":"2024-12-03T03:12:39.062552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Display potential Bar graphs\ndef bar_graph(counts, title):\n    # Plotting the bar chart\n    plt.figure(figsize=(15, 5))\n    plt.bar(counts.index, counts.values)\n    plt.xlabel('Values')\n    plt.ylabel('Count')\n    plt.title(title)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.065089Z","iopub.execute_input":"2024-12-03T03:12:39.065527Z","iopub.status.idle":"2024-12-03T03:12:39.093116Z","shell.execute_reply.started":"2024-12-03T03:12:39.065455Z","shell.execute_reply":"2024-12-03T03:12:39.091904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#a Pie chart of filled vs unfilled\ndef displayPie(table_name_before,table_name_after):\n    empty_count = table_name_before.isna().sum().sum()\n    filled_count = table_name_before.notna().sum().sum()\n    sizes1 = [empty_count, filled_count]\n    \n    labels = [\"empty\", \"filled\"]\n    \n    empty_count_2 = table_name_after.isna().sum().sum()\n    filled_count_2 = table_name_after.notna().sum().sum()\n    sizes2 = [empty_count_2, filled_count_2]\n\n    # Plot the pie charts next to each other\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 6))\n    \n    ax1.pie(sizes1, labels=labels, autopct='%1.1f%%', startangle=140)\n    ax2.pie(sizes2, labels=labels, autopct='%1.1f%%', startangle=140)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.094462Z","iopub.execute_input":"2024-12-03T03:12:39.095006Z","iopub.status.idle":"2024-12-03T03:12:39.109258Z","shell.execute_reply.started":"2024-12-03T03:12:39.094967Z","shell.execute_reply":"2024-12-03T03:12:39.108275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pie chart to display all values\ndef pie_Chart(counts, title):\n    # Plotting the bar chart\n    plt.figure(figsize=(15, 5))\n    plt.pie(counts.index, labels = counts.values)\n    plt.xlabel('Values')\n    plt.ylabel('Count')\n    plt.title(title)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.110628Z","iopub.execute_input":"2024-12-03T03:12:39.111147Z","iopub.status.idle":"2024-12-03T03:12:39.128052Z","shell.execute_reply.started":"2024-12-03T03:12:39.111095Z","shell.execute_reply":"2024-12-03T03:12:39.126909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#NOW create a definition that shows the empty vs filled for EACH sii breakdown. Do this by differing in color?\n\ndef filledFeaturesPlot():\n\n    # Plotting\n    fig, ax = plt.subplots(figsize=(10, 5))\n\n    for i in range(len(tables_to_show)):\n        # Display counts of filled and empty values for each column \n        filled_counts = tables_to_show[i].notna().sum()\n        empty_counts = tables_to_show[i].isna().sum()\n\n        sorted_indices = filled_counts.sort_values(ascending=False).index\n        filled_counts = filled_counts[sorted_indices]\n        empty_counts = empty_counts[sorted_indices]\n\n        # Create stacked horizontal bars\n        ax.barh(sorted_indices, filled_counts, label=f\"{save_name[i]} filled\")\n        ax.barh(sorted_indices, empty_counts, left=filled_counts, label=f\"{save_name[i]} missing\")\n\n    ax.set_xlabel('Count')\n    ax.legend()\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.129484Z","iopub.execute_input":"2024-12-03T03:12:39.129906Z","iopub.status.idle":"2024-12-03T03:12:39.140863Z","shell.execute_reply.started":"2024-12-03T03:12:39.129832Z","shell.execute_reply":"2024-12-03T03:12:39.139642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Display measures for different values\n#Show filled plot for train table\nfilledIndividualPlot(train)\n#Show Age distribution\nbar_graph(train[\"Basic_DemosAge\"].value_counts(), \"Age distribution\")\n#Show sex distribution\nbar_graph(train[\"Basic_DemosSex\"].value_counts(), \"Sex distribution\")\n#Show sii distribution\nbar_graph(train[\"sii\"].value_counts(), \"Sii Distribution\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:39.142107Z","iopub.execute_input":"2024-12-03T03:12:39.142406Z","iopub.status.idle":"2024-12-03T03:12:41.185036Z","shell.execute_reply.started":"2024-12-03T03:12:39.142376Z","shell.execute_reply":"2024-12-03T03:12:41.183930Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCIAT CALCULATIONS AND ACCURACY MEASURES","metadata":{}},{"cell_type":"code","source":"#Measure accuracy of correlations with \"PCIAT\"\ndef pciat_accuracy(table_name):\n    columns = [column for column in train.columns if \"PCIATPCIAT\" in column]\n    PCIAT_table = table_name[columns]\n\n    # Initialize a variable to accumulate the total percentage error\n    total_percentage_error = 0\n    total_count = 0\n\n    # Loop through each row in PCIAT_table\n    for index, row in PCIAT_table.iterrows():\n        predicted_value = row[\"PCIATPCIAT_Total\"]\n        # Skip rows where the predicted value is NaN\n        if pd.isna(predicted_value):\n            continue\n\n        # Calculate the actual value as the sum of columns, excluding \"PCIAT-PCIAT_Total\"\n        actual_value = row.drop(\"PCIATPCIAT_Total\").sum()\n        \n        if pd.isna(actual_value):\n            continue\n        \n        #if the actual value is 0 and so is predicted, we can skip\n        if(actual_value == 0 and predicted_value == 0):\n            total_count += 1\n        elif actual_value == 0:\n            total_percentage_error += 100\n            total_count += 1\n        else:\n            # Calculate the percentage error for the current row\n            row_percentage_error = abs((actual_value - predicted_value) / actual_value) * 100\n            total_percentage_error += row_percentage_error  # Accumulate the error\n            total_count += 1\n\n    # Calculate the mean percentage error across all rows that were measures\n    if total_count == 0:\n        print(\"No PCIAT values were filled\")\n    else:\n        mean_percentage_error = total_percentage_error / total_count\n        print(f\"PCIAT fill error = {mean_percentage_error:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:41.186488Z","iopub.execute_input":"2024-12-03T03:12:41.186910Z","iopub.status.idle":"2024-12-03T03:12:41.195383Z","shell.execute_reply.started":"2024-12-03T03:12:41.186859Z","shell.execute_reply":"2024-12-03T03:12:41.194201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Measure accuracy of correlations with \"PCIAT\"\ndef sii_pciat_accuracy(table_name):\n    # Initialize a variable to accumulate the total percentage error\n    total_matches = 0\n    total_count = 0\n\n    # Loop through each row in PCIAT_table\n    for index, row in table_name.iterrows():\n        predicted_value = row[\"PCIATPCIAT_Total\"]\n        sii_value = row[\"sii\"]\n        # Skip rows where the predicted value is NaN\n        if pd.isna(predicted_value):\n            continue\n        \n        if((predicted_value <= 30 and sii_value == 0.0) or (predicted_value <= 49 and predicted_value >= 31 and sii_value == 1.0) or (predicted_value <= 79 and predicted_value >= 50 and sii_value == 2.0) or (predicted_value <= 100 and predicted_value >= 80 and sii_value == 3.0)):\n            total_matches+=1\n        else:\n            print(predicted_value)\n            print(sii_value)\n            print(index)\n        \n        total_count+=1\n\n    # Calculate the mean percentage error across all rows that were measures\n    mean_percentage_error = total_matches/ total_count * 100\n    print(f\"sii accuracy = {mean_percentage_error:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:41.196917Z","iopub.execute_input":"2024-12-03T03:12:41.197326Z","iopub.status.idle":"2024-12-03T03:12:41.213969Z","shell.execute_reply.started":"2024-12-03T03:12:41.197283Z","shell.execute_reply":"2024-12-03T03:12:41.212768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def PCIAT_row_correlations(filled_table, row):\n    #Apply does a function row by row because axis = 1\n    row_correlations = filled_table.apply(lambda x: abs(row.corr(filled_table)), axis=1)\n    return row_correlations.mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:41.215565Z","iopub.execute_input":"2024-12-03T03:12:41.216065Z","iopub.status.idle":"2024-12-03T03:12:41.229902Z","shell.execute_reply.started":"2024-12-03T03:12:41.216017Z","shell.execute_reply":"2024-12-03T03:12:41.228768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_PCIAT(table_name):\n    #Get the tables that have PCIAT for comparison\n    PCIAT_table = table_name[[column for column in table_name.columns if \"PCIATPCIAT\" in column]]\n\n    #measure current accuracy of sum of all values - to check if the current calculations are correct.\n    pciat_accuracy(table_name)\n    sii_pciat_accuracy(table_name)\n\n    #Fill values for missing PCIAT types if there is only one missing and PCIAT Total is filled and vice versa\n    for index,row in PCIAT_table.iterrows():\n        #Count number of missing values\n        missing_values = row.isna().sum()\n        # Check if only one value is missing in the row\n        if missing_values == 1:\n            #if total is that null value, fill it\n            if(pd.isna(row[\"PCIATPCIAT_Total\"])):\n                row[\"PCIATPCIAT_Total\"] = row.drop(\"PCIATPCIAT_Total\").sum()\n                continue\n            # Calculate the fill value\n            fill_value = row[\"PCIATPCIAT_Total\"] - row.drop(\"PCIATPCIAT_Total\").sum()\n            # Find the column with the missing value and fill it\n            missing_column = row[row.isna()].index[0]\n            PCIAT_table.at[index, missing_column] = fill_value\n            table_name.at[index, missing_column] = fill_value\n\n    #Now fill when there is more than one missing value but total is filled\n    for index,row in PCIAT_table.iterrows():\n        #Get count of missing values AND columns associated with issing counts\n        missing_count = row.drop(\"PCIATPCIAT_Total\").isna().sum()\n        missing_columns = row.index[row.isna()]\n\n        #If total column is filled and its other values having missing values\n        if missing_count>1 and not pd.isna(row[\"PCIATPCIAT_Total\"]):\n            #Find total missing\n            total_val = row[\"PCIATPCIAT_Total\"]\n            filled_sum = row.drop(\"PCIATPCIAT_Total\").dropna().sum()\n            total_val_missing = total_val - filled_sum\n            \n            #If the total is 0, all become 0\n            if total_val == 0:\n                PCIAT_table.loc[index,PCIAT_table.columns] = 0\n                table_name.loc[index,PCIAT_table.columns] = 0\n                continue \n            \n            #calculate min and max sum based on sii\n            min_sum = 0\n            max_sum = 100 - total_val\n            #check if sii is present\n            sii_val = train.loc[index,\"sii\"]\n            if not pd.isna(sii_val):\n                if sii_val == 0.0:\n                    max_sum = 30 - total_val\n                if sii_val == 1.0:\n                    max_sum = 49 - total_val \n                    min_sum = 31 - total_val\n                if sii_val == 2.0:\n                    max_sum = 79 - total_val \n                    min_sum = 50 - total_val \n                if sii_val == 3.0:\n                    min_sum = 80 - total_val\n             \n            #Find ways to add up to this value otherwise\n            possible_lists =[arangement for arangement in product([0,1,2,3,4,5],repeat = missing_count) if sum(arangement)== total_val_missing and sum(arangement)>= min_sum and sum(arangement)<= max_sum]\n            \n            if len(possible_lists)>0:\n                #Now find correlation of these witch other rows! Based on ones with same sii values\n                same_sii  = table_name[table_name[\"sii\"] == table_name.loc[index,\"sii\"]].drop([\"id\", \"sii\"], axis = 1)\n                largest_corr = 0\n                use_combo = possible_lists[0]\n                for arr in possible_lists:\n                    for i in range (len(missing_columns)):\n                        #if only one possible combo, then skip following steps to just fill\n                        if(len(possible_lists)== 1):\n                            use_combo = arr\n                        else:\n                            same_sii = same_sii[same_sii[missing_columns[i]] == arr[i]]\n                            corr = PCIAT_row_correlations(same_sii, table_name.drop([\"id\", \"sii\"], axis = 1).loc[index,:])\n                            if(corr >= largest_corr):\n                                use_combo = arr     \n\n                #now fill missing values\n                for i in range (len(missing_columns)):\n                    table_name.at[index,missing_columns[i]] = use_combo[i]\n                    PCIAT_table.at[index,missing_columns[i]] = use_combo[i]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:41.231373Z","iopub.execute_input":"2024-12-03T03:12:41.231865Z","iopub.status.idle":"2024-12-03T03:12:41.253280Z","shell.execute_reply.started":"2024-12-03T03:12:41.231779Z","shell.execute_reply":"2024-12-03T03:12:41.252057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Overhere, we are trying to fill in all the missing PCIAT values. \nWe know the following:\n- PCIAT total = total of all 20 values\n    - Represent: Severity Impairment Index: 0-30=None; 31-49=Mild; 50-79=Moderate; 80-100=Severe\n1. First we fill in all values that are missing just ONE value\n2. Now we fill in total values! This is done twice but for good reason - we need rows that are fully filled for the next steps.\n3. We try to fill in missing values if total is present and there are more than 1 other values missing utilizing probability:\n    - #x1+x2+x3 = y\n    - x1, x2, x3, x4 = 1|2|3|4\n    - What are all the possibile arrangements? unordered with replacement!\n    - Now given each arrangement and all the ALREADY filled rows, what is the accuracy?\n        - We measure the accuracy by correlation. \n        - So first we get the indexes of ALL rows that are already filled with the same predictions.\n        - now we get the correlation between this row and all other rows by doing a simple .corr!\n    - Choose the combination that gives the HIGHEST correlation!\n3. now we once again fill in all total values if the rest of the data is filled!\n4. If total is unfilled and so is values, we leave it up to correlation :)","metadata":{}},{"cell_type":"code","source":"fill_PCIAT(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:41.254819Z","iopub.execute_input":"2024-12-03T03:12:41.255317Z","iopub.status.idle":"2024-12-03T03:12:44.909855Z","shell.execute_reply.started":"2024-12-03T03:12:41.255270Z","shell.execute_reply":"2024-12-03T03:12:44.908694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:44.911128Z","iopub.execute_input":"2024-12-03T03:12:44.911476Z","iopub.status.idle":"2024-12-03T03:12:45.088422Z","shell.execute_reply.started":"2024-12-03T03:12:44.911441Z","shell.execute_reply":"2024-12-03T03:12:45.087340Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FUNCTIONS TO IMPUTE MISSING VALUES VIA mean, median, or mode","metadata":{}},{"cell_type":"code","source":"#function to fill in mean median mode\ndef imputeValues(df, feature, table_name, data_type):\n    #fill missing values based on feature type! int = mean, categorical int = mode\n    if(data_type == \"Continuous\"):\n        df.loc[:, feature] = df[feature].fillna(table_name[feature].mean())\n    if(data_type == \"Categorical\"):\n        df.loc[:, feature] = df[feature].fillna(table_name[feature].mode().iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.089992Z","iopub.execute_input":"2024-12-03T03:12:45.090421Z","iopub.status.idle":"2024-12-03T03:12:45.097975Z","shell.execute_reply.started":"2024-12-03T03:12:45.090373Z","shell.execute_reply":"2024-12-03T03:12:45.096599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Is this working?","metadata":{}},{"cell_type":"code","source":"#function for imputing data\ndef lessThen5(table_name, disinclude_columns = [\"sii\"]):\n    for column_name in table_name.columns:\n        #PCIAT is imputed in a different manner!\n        if(\"PCIATPCIAT\" in column_name or \"id\" in column_name):\n            continue\n            \n        missing_percentage = table_name[column_name].isna().mean() * 100\n        # Check if missing data is less than 5%\n        if missing_percentage <= 5 and missing_percentage>0 and column_name not in disinclude_columns:\n            data_type = get_data_type(column_name)\n            imputeValues(table_name, column_name, table_name, data_type)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.099355Z","iopub.execute_input":"2024-12-03T03:12:45.099775Z","iopub.status.idle":"2024-12-03T03:12:45.116289Z","shell.execute_reply.started":"2024-12-03T03:12:45.099705Z","shell.execute_reply":"2024-12-03T03:12:45.115103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lessThen5(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.118168Z","iopub.execute_input":"2024-12-03T03:12:45.118607Z","iopub.status.idle":"2024-12-03T03:12:45.140109Z","shell.execute_reply.started":"2024-12-03T03:12:45.118558Z","shell.execute_reply":"2024-12-03T03:12:45.138702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.141682Z","iopub.execute_input":"2024-12-03T03:12:45.142152Z","iopub.status.idle":"2024-12-03T03:12:45.318972Z","shell.execute_reply.started":"2024-12-03T03:12:45.142104Z","shell.execute_reply":"2024-12-03T03:12:45.317659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Functions used to drop outliers based on entropies","metadata":{}},{"cell_type":"code","source":"def entropyCalcJaccard(table_name):\n    for column in table_name.columns:\n        table_name[column] = pd.to_numeric(table_name[column], errors='coerce')\n\n    # Calculate pairwise Jaccard similarity\n    jaccard_similarity = pdist(table_name, metric = 'jaccard')\n    similarity_matrix = 1 - squareform(jaccard_similarity)\n\n    # Calculate entropy for each row based on similarity to other rows\n    row_entropies = entr(similarity_matrix).sum(axis=1)/np.log(2)\n    \n    return row_entropies","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.326016Z","iopub.execute_input":"2024-12-03T03:12:45.326387Z","iopub.status.idle":"2024-12-03T03:12:45.332316Z","shell.execute_reply.started":"2024-12-03T03:12:45.326353Z","shell.execute_reply":"2024-12-03T03:12:45.331066Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Compute a similarity measure between each pair of rows\n    - Jaccard Similarity\n    - measures the similarity between sample sets by measuring overlap\n    - Good for sparse data\n    - we want to SMOOTH because the data is so sparse! LAPACE SMOOTHING\n2. The result is a NORMALIZED matrix that displays similarities between rows i with size [i,j]\n3. We shouldnt normalize further due to the nature of sparse data\n4. NOW calculate the entropy of each row via shannon entropy!\n    - Shannon Entropy: basically calculates entropy of each individual value\n        - H(X) is entropy of the dataset\n        - p(Xi) is probability of observing value Xi in data\n        - H(X) = -sum(p(Xi)*log2(p(Xi)))\n5. This method allows for entropy calculation of rows AFTER we already know measures of similarity to other rows.\n\n   //TODO Do we need to normalize columns? or Normalize rows to add up to one! this allows for accurate entropy calculation because we need PROBABILITY distribution using z-score by column MIGHT defeat the clustering goal of going until stdev is <=1 for entropies  <=1 for entropies","metadata":{}},{"cell_type":"code","source":"#calculate entropy of cluster sent in as a table - returns entropy of each row as a list. \ndef entropyCalc(table_name):\n    for column in table_name.columns:\n        table_name[column] = pd.to_numeric(table_name[column], errors='coerce')\n\n    # Calculate pairwise similarity (e.g., using cosine similarity)\n    similarity_matrix = 1 - cdist(table_name, table_name, metric='cosine')\n\n    # Calculate entropy for each row based on similarity to other rows\n    row_entropies = entr(similarity_matrix).sum(axis=1)/np.log(2)\n    \n    return row_entropies","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.333619Z","iopub.execute_input":"2024-12-03T03:12:45.333968Z","iopub.status.idle":"2024-12-03T03:12:45.346042Z","shell.execute_reply.started":"2024-12-03T03:12:45.333937Z","shell.execute_reply":"2024-12-03T03:12:45.344900Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Compute a similarity measure between each pair of rows\n    - Cosine similarity!\n    - measures the similarity between two vectors of an inner product space. \n    - measured by the cosine of the angle between two vectors and determines whether two vectors are pointing in roughly the same direction\n    - NOT the best for sparse data\n3. The result is a matrix that displays similarities between rows i with size [i,j]\n5. Calculate the entropy of each row via shannon entropy!\n    - Shannon Entropy: basically calculates entropy of each individual value\n        - H(X) is entropy of the dataset\n        - p(Xi) is probability of observing value Xi in data\n        - H(X) = -sum(p(Xi)*log2(p(Xi)))\n6. This method allows for entropy calculation of rows AFTER we already know measures of similarity to other rows.","metadata":{}},{"cell_type":"code","source":"#The table_name is what we want to split based on percentiles until stdev is <1 for the entropies in any cluster, at which point we add to the final_cluster.\ndef splitClustersJaccard(table_name, final_cluster):\n    \n    #First calculate entropy of all rows!\n    entropy = entropyCalcJaccard(table_name)\n\n    #If stdev is good, then just return! No need to split\n    stdev = np.std(entropy)\n    if(abs(stdev) <= 1):\n        final_cluster.append(table_name.index)\n\n    #otherwise cluster until stdev is good!\n    else:\n        \n        #NOW get the deviations! Via frequencies\n        splits = (max(entropy) - min(entropy))/4\n        cluster_split = [(splits*1)+min(entropy), (splits*2)+min(entropy), (splits*3)+min(entropy), max(entropy)]\n\n        splitA = []\n        splitB = []\n        splitC = []\n        splitD = []\n\n        for i, (idx, row) in enumerate(table_name.iterrows()):\n            if entropy[i] <= cluster_split[0]:\n                splitA.append(idx)\n            elif entropy[i] <= cluster_split[1]:\n                splitB.append(idx)\n            elif entropy[i] <= cluster_split[2]:\n                splitC.append(idx)\n            else:\n                splitD.append(idx)\n\n        splitA = table_name.loc[splitA, :]\n        splitB = table_name.loc[splitB, :]\n        splitC = table_name.loc[splitC, :]\n        splitD = table_name.loc[splitD, :]\n\n        #Start recurrsion! ONLY if there actually is data to split though\n        if not splitA.empty:\n            splitClustersJaccard(splitA, final_cluster)\n        if not splitB.empty:\n            splitClustersJaccard(splitB, final_cluster)\n        if not splitC.empty:\n            splitClustersJaccard(splitC, final_cluster)\n        if not splitD.empty:\n            splitClustersJaccard(splitD, final_cluster)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.347376Z","iopub.execute_input":"2024-12-03T03:12:45.347735Z","iopub.status.idle":"2024-12-03T03:12:45.366782Z","shell.execute_reply.started":"2024-12-03T03:12:45.347700Z","shell.execute_reply":"2024-12-03T03:12:45.365667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The table_name is what we want to split based on percentiles until stdev is <1 for the entropies in any cluster, at which point we add to the final_cluster.\ndef splitClustersCosine(table_name, final_cluster):\n    \n    #First calculate entropy of all rows!\n    entropy = entropyCalc(table_name)\n\n    #If stdev is good, then just return! No need to split\n    stdev = np.std(entropy)\n    if(abs(stdev) <= 0.1):\n        final_cluster.append(table_name.index)\n\n    #otherwise cluster until stdev is good!\n    else:\n        \n        #NOW get the deviations! Via percentile\n        cluster_split = [np.percentile(entropy,25), np.percentile(entropy,50), np.percentile(entropy,75), max(entropy)]\n\n        splitA = []\n        splitB = []\n        splitC = []\n        splitD = []\n\n        for i, (idx, row) in enumerate(table_name.iterrows()):\n            if entropy[i] <= cluster_split[0]:\n                splitA.append(idx)\n            elif entropy[i] <= cluster_split[1]:\n                splitB.append(idx)\n            elif entropy[i] <= cluster_split[2]:\n                splitC.append(idx)\n            else:\n                splitD.append(idx)\n\n        splitA = table_name.loc[splitA, :]\n        splitB = table_name.loc[splitB, :]\n        splitC = table_name.loc[splitC, :]\n        splitD = table_name.loc[splitD, :]\n\n        #Start recurrsion! ONLY if there actually is data to split though\n        if not splitA.empty:\n            splitClustersCosine(splitA, final_cluster)\n        if not splitB.empty:\n            splitClustersCosine(splitB, final_cluster)\n        if not splitC.empty:\n            splitClustersCosine(splitC, final_cluster)\n        if not splitD.empty:\n            splitClustersCosine(splitD, final_cluster)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.368417Z","iopub.execute_input":"2024-12-03T03:12:45.368906Z","iopub.status.idle":"2024-12-03T03:12:45.383760Z","shell.execute_reply.started":"2024-12-03T03:12:45.368833Z","shell.execute_reply":"2024-12-03T03:12:45.382574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The recursive method above:\n1. Calculate the entropy of all rows within the cluster!\n    - Low entropy = the cluster is highly homogenious and pure!\n    - high entropy = NOT very homogenious. Could entail outliers...\n2. Check if the current cluster as entropy standard deviation of less than 1.\n    - This represents a decent distribution!\n    - At this point add the indexes of the values in the cluster to the Final_cluster 2-D list\n4. OTHERWISE clustering by distributions of 4!\n    - group the rows into 4 tables, representing clusters to further evaluate\n    - send each cluster back to step 2 to improve!","metadata":{}},{"cell_type":"code","source":"def convertToCategorical(table_name,columnName):\n    new_column = []\n    divide_by = np.std(table_name[columnName])\n    for value in table_name[columnName].values:\n        if pd.isna(value):\n            new_column.append(0)\n        else:\n            new_column.append(math.ceil(value/divide_by))\n        \n    return pd.DataFrame(new_column)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.385003Z","iopub.execute_input":"2024-12-03T03:12:45.385347Z","iopub.status.idle":"2024-12-03T03:12:45.399396Z","shell.execute_reply.started":"2024-12-03T03:12:45.385316Z","shell.execute_reply":"2024-12-03T03:12:45.398394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#NORMALIZE the columns!\ndef normalize(table):\n    # Create a MinMaxScaler object\n    scaler = MinMaxScaler()\n    df_normalized_zscore = pd.DataFrame(scaler.fit_transform(table), columns=table.columns, index=table.index)\n    return df_normalized_zscore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.401037Z","iopub.execute_input":"2024-12-03T03:12:45.401356Z","iopub.status.idle":"2024-12-03T03:12:45.421714Z","shell.execute_reply.started":"2024-12-03T03:12:45.401324Z","shell.execute_reply":"2024-12-03T03:12:45.420321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clusterEntropy(table_name, method = None):\n    #First convert entire table to categorical values\n    use_table = table_name.copy()\n    \n    #Second make sure to fill in all blank values with 0!\n    use_table = use_table.fillna(0)\n\n    all_clusters = []\n\n    for column in table_name.columns:\n        if(get_data_type(column) == \"Continuous\"):\n            use_table[column] = convertToCategorical(table_name,column)\n\n    if(method == \"Cosine\"):\n        #Normalize for Cosine\n        use_table = normalize(use_table)\n\n        #clusters will return a list of clusters containing index values\n        splitClustersCosine(use_table, all_clusters)\n\n    elif(method == \"Jaccard\"):\n        #clusters will return a list of clusters containing index values\n        splitClustersJaccard(use_table, all_clusters)\n    \n    all_clusters.sort(key = lambda x: len(x))\n\n    droppingClusters = []\n    for cluster in all_clusters:\n        #print(table_name.loc[cluster])\n        #Handle single index clusters, marked as OUTLIER\n        if (len(cluster) == 1):\n            droppingClusters.append(cluster[0])\n        #IF has at least 3 values, mayhaps we can impute missing values of column!\n        elif(len(cluster)>2):\n            print(\"Imputing!\") \n            \n            for column_name in table_name.columns:\n                #PCIAT is imputed in a different manner!\n                if(\"PCIATPCIAT\" in column_name or \"id\" in column_name):\n                    continue\n                    \n                missing_percentage = table_name.loc[cluster,column_name].isna().mean() * 100\n                # Check if missing data is less than 5%\n                if missing_percentage <= 5 and missing_percentage>0 :\n                    \n                    data_type = get_data_type(column_name)\n                    \n                    if(data_type == \"Continuous\"):\n                        \n                        subset = table_name.loc[cluster, column_name]\n                        fillVal = subset.mean() \n                        fillVal = table_name[column_name].dtype.type(fillVal)\n                        subset = subset.fillna(fillVal)\n                        table_name.loc[cluster, column_name] = subset\n                        print(f\"Filling {column_name} to be {fillVal}\")\n                        \n                    if(data_type == \"Categorical\"):\n                        \n                        subset = table_name.loc[cluster, column_name]\n                        fillVal = subset.mode().iloc[0]\n                        fillVal = table_name[column_name].dtype.type(fillVal)\n                        subset = subset.fillna(fillVal)\n                        table_name.loc[cluster, column_name] = subset\n                        print(f\"Filling {column_name} to be {fillVal}\")\n                        \n            print(table_name.loc[cluster])\n        else:\n            print(table_name.loc[cluster])\n        \n\n    return droppingClusters","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.423309Z","iopub.execute_input":"2024-12-03T03:12:45.423667Z","iopub.status.idle":"2024-12-03T03:12:45.438689Z","shell.execute_reply.started":"2024-12-03T03:12:45.423633Z","shell.execute_reply":"2024-12-03T03:12:45.437460Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"An Idea: using clustering AND calculating Entropy\n- Clustering groups data into 4 clusters based on entropies of rows UNTIL the STDEV of entropies within any cluster is less than 0.5 - because there are not many outliers in this data!\n1. first we need to deal with pre-processing the data.\n    - missing values: What if we just filled all missing values to be temporarily 0 - to account for sparsity!\n    - We need to convert all continuous data into categorical data!\n        - Do this by splitting each by STDEV to allow us to keep standard of variability\n    - We need to make sure all row values add up to one! Aka. NORMALIZE the data\n        - Doing this allows us to work with imbalanced data, because the spread is not a uniform distribution \n2. Send into recursive method to calculate optimal clusters and get the clusters to be full of index clusters \n3. After reaching this point, deal with outliers\n    - \"clusters\" with just one index is considered an OUTLIER. they fit in no where else. DROP \n    - small clusters may ALL be outliers, especially if most have 0 values.\n        1. take these and potentially try to fill in missing values accordingly? \n            - Drop all 0s \n            - Run the clustering algorithm using rows that have the same number of filled features\n            - Fill in missing values based on rows that are similarly related? Maybe?\nBenefits:\n    - Entropy allows for better measure when there is imbalanced data!\n    - Entropy is better for categorical variables, because distance calculations rely more on continuous\n    - Works with sparse data","metadata":{}},{"cell_type":"code","source":"print(\"Before\")\ntrain.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.440280Z","iopub.execute_input":"2024-12-03T03:12:45.440608Z","iopub.status.idle":"2024-12-03T03:12:45.627040Z","shell.execute_reply.started":"2024-12-03T03:12:45.440576Z","shell.execute_reply":"2024-12-03T03:12:45.625930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sii_1 = train[train['sii'] == 0]\nsii_2 = train[train['sii'] == 1]\nsii_3 = train[train['sii'] == 2]\nsii_4 = train[train['sii'] == 3]\n\n\nsii_1.drop(['sii', 'id'], axis = 1, inplace = True)\nsii_2.drop(['sii', 'id'], axis = 1, inplace = True)\nsii_3.drop(['sii', 'id'], axis = 1, inplace = True)\nsii_4.drop(['sii', 'id'], axis = 1, inplace = True)\n\ndrop_clusters_1 = clusterEntropy(sii_1, method = \"Jaccard\")\ndrop_clusters_2 = clusterEntropy(sii_2, method = \"Jaccard\")\ndrop_clusters_3 = clusterEntropy(sii_3, method = \"Jaccard\")\ndrop_clusters_4 = clusterEntropy(sii_4, method = \"Jaccard\")\n\ndrop_clusters_jaccard = drop_clusters_1 + drop_clusters_2 + drop_clusters_3 + drop_clusters_4\nprint(drop_clusters_jaccard)\n\ndrop_clusters_cosine = clusterEntropy(train, method = \"Cosine\")\nprint(drop_clusters_cosine)\n\nprint(set(drop_clusters_jaccard).intersection(set(drop_clusters_cosine)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:12:45.628665Z","iopub.execute_input":"2024-12-03T03:12:45.629131Z","iopub.status.idle":"2024-12-03T03:13:43.485663Z","shell.execute_reply.started":"2024-12-03T03:12:45.629083Z","shell.execute_reply":"2024-12-03T03:13:43.484573Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Test with Jaccard\ntrain.drop(drop_clusters_jaccard, inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:43.486947Z","iopub.execute_input":"2024-12-03T03:13:43.487274Z","iopub.status.idle":"2024-12-03T03:13:43.495973Z","shell.execute_reply.started":"2024-12-03T03:13:43.487242Z","shell.execute_reply":"2024-12-03T03:13:43.494924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Test with Cosine\n#df.drop(drop_clusters_cosine, inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:43.497391Z","iopub.execute_input":"2024-12-03T03:13:43.497741Z","iopub.status.idle":"2024-12-03T03:13:43.510538Z","shell.execute_reply.started":"2024-12-03T03:13:43.497704Z","shell.execute_reply":"2024-12-03T03:13:43.509390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:43.512055Z","iopub.execute_input":"2024-12-03T03:13:43.512434Z","iopub.status.idle":"2024-12-03T03:13:43.698997Z","shell.execute_reply.started":"2024-12-03T03:13:43.512401Z","shell.execute_reply":"2024-12-03T03:13:43.697930Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SPLIT TABLE by Sii and re-do calculations!","metadata":{}},{"cell_type":"code","source":"#Get ONLY the data that has SII\nnone_table = train[train['sii'] == 0.0]\nmild_table = train[train['sii'] == 1.0]\nmoderate_table = train[train['sii'] == 2.0]\nsevere_table = train[train['sii'] == 3.0]\n\n#this is used going forward in multiple functions\ntables_to_show = [none_table, mild_table, moderate_table, severe_table]\nsave_name = [\"none\", \"mild\", \"moderate\", \"severe\", \"train\"]\nsii_correlations = {}\n\n#Show filled for each SII\nfilledFeaturesPlot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:43.700566Z","iopub.execute_input":"2024-12-03T03:13:43.701052Z","iopub.status.idle":"2024-12-03T03:13:45.629039Z","shell.execute_reply.started":"2024-12-03T03:13:43.700995Z","shell.execute_reply":"2024-12-03T03:13:45.627916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Fill all missing values and drop outliers\nfor tables in tables_to_show:\n    lessThen5(tables)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:45.630386Z","iopub.execute_input":"2024-12-03T03:13:45.630741Z","iopub.status.idle":"2024-12-03T03:13:45.686768Z","shell.execute_reply.started":"2024-12-03T03:13:45.630706Z","shell.execute_reply":"2024-12-03T03:13:45.685864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calculate Correlations","metadata":{}},{"cell_type":"code","source":"def corrCatToCon(table_name, columnVal, columnCompute,dataVal, dataCompute):\n\n    if dataVal == \"Categorical\":\n        # Point-biserial correlation for binary categorical columnVal\n        if table_name[columnVal].nunique() == 2:\n            le = preprocessing.LabelEncoder()\n            encoded_columnVal = le.fit_transform(table_name[columnVal])\n            corr, p = stats.pointbiserialr(encoded_columnVal, table_name[columnCompute])\n            if p <= 0.05:\n                if abs(corr)>=0.7:\n                    return corr, \"Strong\"\n                elif abs(corr)>=0.3:\n                    return corr, \"Medium\"\n                elif abs(corr)>=0.1:\n                    return corr, \"Weak\"\n                else:\n                    return corr, \"Nada\"\n        # ANOVA for multi-level categorical columnVal\n        else:\n            check_enough = table_name[table_name[columnVal].notna() & table_name[columnCompute].notna()]\n            if len(check_enough[columnVal].value_counts())<2 or check_enough.empty:\n                return 0, \"Nada\"\n            model = ols(f'{columnCompute} ~ C({columnVal})', data=check_enough).fit()\n            try:\n                anova_table = sm.stats.anova_lm(model, typ=2)\n            except Exception as e:\n                print(f\"Issue within {columnVal} and {columnCompute}: {e}\")\n                return 0, \"Nada\"\n            else:\n                p_value = anova_table['PR(>F)'][f\"C({columnVal})\"]\n\n                # Calculate Eta-squared\n                SS_between = anova_table['sum_sq'][f\"C({columnVal})\"]\n                SS_total = anova_table['sum_sq'].sum()\n                eta_squared = SS_between / SS_total\n            \n                if p_value <= 0.05:\n                    if eta_squared >=0.14:\n                        return eta_squared, \"Strong\"\n                    elif eta_squared >=0.06:\n                        return eta_squared, \"Medium\"\n                    else:\n                        return eta_squared, \"Weak\"\n\n    elif dataCompute == \"Categorical\":\n        # Point-biserial correlation for binary categorical columnCompute\n        if table_name[columnCompute].nunique() == 2:\n            le = preprocessing.LabelEncoder()\n            encoded_columnCompute = le.fit_transform(table_name[columnCompute])\n            corr, p = stats.pointbiserialr(encoded_columnCompute, table_name[columnVal])\n            if p <= 0.05:\n                if abs(corr)>=0.7:\n                    return corr, \"Strong\"\n                elif abs(corr)>=0.3:\n                    return corr, \"Medium\"\n                elif abs(corr)>=0.1:\n                    return corr, \"Weak\"\n                else:\n                    return corr, \"Nada\"\n        # ANOVA for multi-level categorical columnCompute\n        else:\n            check_enough = table_name[table_name[columnVal].notna() & table_name[columnCompute].notna()]\n            if len(check_enough[columnCompute].value_counts())<2 or check_enough.empty:\n                return 0, \"Nada\"\n            model = ols(f'{columnVal} ~ C({columnCompute})', data=check_enough).fit()\n            try:\n                anova_table = sm.stats.anova_lm(model, typ=2)\n            except Exception as e:\n                print(f\"Issue within {columnVal} and {columnCompute}: {e}\")\n                return 0, \"Nada\"\n            else:\n                p_value = anova_table['PR(>F)'][f\"C({columnCompute})\"]\n\n                # Calculate Eta-squared\n                SS_between = anova_table['sum_sq'][f\"C({columnCompute})\"]\n                SS_total = anova_table['sum_sq'].sum()\n                eta_squared = SS_between / SS_total\n            \n                if p_value <= 0.05:\n                    if eta_squared >=0.14:\n                        return eta_squared, \"Strong\"\n                    elif eta_squared >=0.06:\n                        return eta_squared, \"Medium\"\n                    else:\n                        return eta_squared, \"Weak\"\n\n    return 0, \"Nada\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:24:59.813556Z","iopub.execute_input":"2024-12-03T03:24:59.814001Z","iopub.status.idle":"2024-12-03T03:24:59.829993Z","shell.execute_reply.started":"2024-12-03T03:24:59.813959Z","shell.execute_reply":"2024-12-03T03:24:59.828672Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Use Point biseral to calculate correlation f columnVal is categorical and only has 2 values\n    - P must be <= 0.05 to have a correlation\n    - corr must be >= 0.7 to be significant\n2. use ANOVA otherwise\n    - P must be <= 0.05 to have correlation\n    - variance must be >= 0.14 to be significant \n    - compares the mean values of the continuous variable across different categories of the categorical variable\n    - scaled to match min/max values of other coparisons via\n        - scaled_eta = 43.75* eta_squared * eta_squared - (6.75 * eta_squared) +0.3375 \n    - Type 2: \n        - Tests each main effect after accounting for all other main effects, but without considering interactions.\n        - Useful when you have no interactions in the model and want to understand the unique contribution of each factor.\n        - Often preferred for balanced designs (where each group has an equal number of observations).\n    - Checks:\n        - The data needs to be normally distributed.\n        - The data is distributed with equal variance.\n        - There are no drastic outliers in the data.\n        - The groups are independent of each other.\n    - sum of squares between groups -the total sum of squares.","metadata":{}},{"cell_type":"code","source":"#Return column Compute if it is a highly correlated column!\ndef PearsonCorrelation(table_name, columnVal, columnCompute):\n    \n    # Ensure variance > 0 for both\n    if table_name[columnCompute].var() == 0 or table_name[columnVal].var() == 0:\n        print(f\"Not enough variance within {columnCompute} or {columnVal}\")\n        return 0, \"Nada\"\n\n    check_enough = table_name[[columnVal, columnCompute]].dropna()\n    if(check_enough.shape[0] < 13):\n        print(f\"Not enough data points between {columnCompute} or {columnVal}\")\n        return 0, \"Nada\"\n\n    #Check for correlations   \n    with warnings.catch_warnings():\n        warnings.simplefilter(\"error\", RuntimeWarning)\n        try:\n            correlation = table_name[columnVal].corr(table_name[columnCompute])\n            if abs(correlation)>=0.7:\n                return correlation, \"Strong\"\n            elif abs(correlation)>=0.3:\n                return correlation, \"Medium\"\n            elif abs(correlation)>=0.1:\n                return correlation, \"Weak\"\n            else:\n                return correlation, \"Nada\"\n        except Exception as e:\n            print(f\"Correlation between {columnCompute} and {columnVal} is probably 0\")\n            return 0, \"Nada\"\n            \n    return 0, \"Nada\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:45.705038Z","iopub.execute_input":"2024-12-03T03:13:45.705466Z","iopub.status.idle":"2024-12-03T03:13:45.720061Z","shell.execute_reply.started":"2024-12-03T03:13:45.705418Z","shell.execute_reply":"2024-12-03T03:13:45.718940Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* The above function that will calculate the correlations between two columns - a PEARSON correlation\n    * it will return nothing if the columns are NOT correlated, but it will return the correlation if they ARE\n* Things to accoutn for:\n    * Ensure columnCompute has variance > 0 - otherwise skip\n    * Ensure there are enough data points for comparison - columns should ALWAYS have>2 to compare\n    * if corr > 0.7, it is considered a strong correlation!\n    \nWhat is a pearson correlation? It returns a value between -1 and 1 regardless of order!\n- r = [n(Σxy) - (Σx)(Σy)] / √[n(Σx²) - (Σx)²][n(Σy²) - (Σy)²]\n    - r: is the Pearson correlation coefficient\n    - n: is the sample size\n    - Σ: represents summation\n    - x: is the independent variable\n    - y: is the dependent variable\n    - Why does it work?\n        - The numerator gives us the directionality by suming up the product of each standard deviation at the same index\n        - The denominator is just the numerator squared AND THEN sqrt. \n            - this is to normalize the data between -1 and 1\n            - MAKING SURE the value is positive as to not affect the directinality found in the numerator\n- r^2 = (coefficient of determination) to understand the proportion of variance explained","metadata":{}},{"cell_type":"code","source":"def chi_squared(table_name, columnVal, columnCompute):\n    \n    check_enough = table_name[[columnVal, columnCompute]].dropna()\n    if(check_enough.shape[0] < 13):\n        #Insert the Barnards exact test - designed for 2 x 2 though\n        print(f\"Not enough data points between {columnCompute} and {columnVal}\")\n        return 0, \"Nada\"\n    \n    contingency_table = pd.crosstab(check_enough[columnVal], check_enough[columnCompute])\n    \n    #get chisquared\n    chi2, p, dof, expected = chi2_contingency(contingency_table)\n    \n    #If frequency less than 5, then we cant say anything for certain\n    if (expected < 5).any():\n        #Try a different test?\n        print(f\"Expected value is less then 5 between {columnCompute} and {columnVal}\")\n        return 0, \"Nada\"\n    \n    #now get correlation coefficient\n    columnValCounts = check_enough[columnVal].value_counts().shape[0]\n    columnComCounts = check_enough[columnCompute].value_counts().shape[0]\n    minCount = min(columnValCounts, columnComCounts)-1\n    if(minCount == 0):\n        print(f\"Either {columnCompute} or {columnVal} only had one value\")\n        return 0, \"Nada\"\n    den = minCount*check_enough.shape[0]\n    \n    cramer_v = (chi2/den) ** 0.5     \n    \n    #If chi_squared <= 0.05 \n    if p <= 0.05:\n        if dof==1:\n            if cramer_v>=0.5:\n                return cramer_v, \"Strong\"\n            elif cramer_v>=0.3:\n                return cramer_v, \"Medium\"\n            else:\n                return cramer_v, \"Weak\"\n        elif dof == 2:\n            if cramer_v>=0.35:\n                return cramer_v, \"Strong\"\n            elif cramer_v>=0.21:\n                return cramer_v, \"Medium\"\n            else:\n                return cramer_v, \"Weak\"\n        elif dof == 3:\n            if cramer_v>=0.29:\n                return cramer_v, \"Strong\"\n            elif cramer_v>=0.17:\n                return cramer_v, \"Medium\"\n            else:\n                return cramer_v, \"Weak\"\n        else:\n            if cramer_v>=0.22:\n                return cramer_v, \"Strong\"\n            elif cramer_v>=0.13:\n                return cramer_v, \"Medium\"\n            else:\n                return cramer_v, \"Weak\"\n    \n    return 0, \"Nada\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:45.721475Z","iopub.execute_input":"2024-12-03T03:13:45.721811Z","iopub.status.idle":"2024-12-03T03:13:45.738554Z","shell.execute_reply.started":"2024-12-03T03:13:45.721778Z","shell.execute_reply":"2024-12-03T03:13:45.737367Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We use the Chi-Squared test to find correlations between categorical variable.\n- Chi-Square measures how much the observed frequencies in a contingency table deviate from expected frequencies under the null hypothesis of independence.\n    - Contingency Table:\n    - null Hypothesis: that the columns are indepemdant of each other\n    - The Chi-square statistic increases with the size of the contingency table. \n1. Checks:\n    - chi-Squared test must have at least 5 frequencies\n    - recommended 13 samples\n    - p val must be <=0.05 to be significant!\n    - crammer rule gives value. Strong correlation measures change based on degrees of freedom\n        - min(k-1, r-1) is used to adjust for the smaller dimension of the contingency table ensuring that Cramér's V is normalized regardless of table size\n        - used since chi-squared increases with size, and we want a number to represent correlation between 0 and 1\n        - For \n            1. df=1:\t- 0.10: Small\n                        - 0.30: Medium\n                        - 0.50: Large\n            2. df=2:\t- 0.07: Small\n                        - 0.21: Medium\n                        - 0.35: Large\n            3. df=3:\t- 0.06: Small\n                        - 0.17: Medium\n                        - 0.29: Large\n            4. df≥4:\t- 0.05: Small\n                        - 0.13: Medium\n                        - 0.22: Large","metadata":{}},{"cell_type":"code","source":"#Displays all correlations\ndef heat_map(data):\n    plt.figure(figsize=(10, 10))\n    \n    # plotting the heatmap \n    hm = sn.heatmap(data = data, vmin = 0, vmax = 1) \n  \n    # displaying the plotted heatmap \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:29:47.613128Z","iopub.execute_input":"2024-12-03T03:29:47.614046Z","iopub.status.idle":"2024-12-03T03:29:47.619832Z","shell.execute_reply.started":"2024-12-03T03:29:47.614002Z","shell.execute_reply":"2024-12-03T03:29:47.618597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def displayCorr(table):\n\n    size = len(table.columns)\n    data = np.empty((size,size))\n    strength= None\n    finalCorr = None\n\n    for x, columnA in enumerate(table.columns):\n        vals1 = get_data_type(columnA)\n        for y, columnB in enumerate(table.columns):\n            vals2 = get_data_type(columnB)\n                \n            if(columnA==columnB):\n                data[x,y] = 1\n                continue\n    \n            if (vals1== vals2):\n                #If both categorical use chi!\n                if(vals1 == \"Categorical\"):\n                    finalCorr, strength = chi_squared(table, columnA, columnB)\n                #If both Continuous use pearson!\n                else:\n                    finalCorr, strength = PearsonCorrelation(table, columnA, columnB)\n            #If one ctaegorical and the other continuous, use the other one\n            else:\n                finalCorr, strength = corrCatToCon(table, columnA, columnB, vals1, vals2)\n\n            if strength == None:\n                data[x,y] = 0\n            elif strength == \"Strong\":\n                data[x,y] = 0.8\n            elif strength == \"Medium\":\n                data[x,y] = 0.6\n            elif strength == \"Weak\":\n                data[x,y] = 0.4\n            else:\n                data[x,y] = 0.2\n                \n    table_corr = pd.DataFrame(data, columns = table.columns, index = table.columns)\n    heat_map(table_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:27:16.616733Z","iopub.execute_input":"2024-12-03T03:27:16.617148Z","iopub.status.idle":"2024-12-03T03:27:16.626588Z","shell.execute_reply.started":"2024-12-03T03:27:16.617112Z","shell.execute_reply":"2024-12-03T03:27:16.625413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#displayCorr(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:13:45.770352Z","iopub.execute_input":"2024-12-03T03:13:45.770755Z","iopub.status.idle":"2024-12-03T03:13:45.785346Z","shell.execute_reply.started":"2024-12-03T03:13:45.770722Z","shell.execute_reply":"2024-12-03T03:13:45.784364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recalibrateCorrelations(table_name, columnName = None, i = None):\n    \n    #We do not want to correlate to the ID\n    table_name = table_name.drop(columns=[\"id\"])\n    \n    correlations = {}\n    \n    #Send in every columnc columnVal to find its best correlated columns\n    for columnVal in table_name.columns:\n        \n        #Initialize dictionary of columns highly correlated to columnVal\n        correlations[columnVal] = []\n        \n        #Get the data type for the column we are attemtping too understand!\n        data_type_columnVal = get_data_type(columnVal)\n        \n        #Check the specific column correlation, adding or removing accordingly:\n        if columnName != None and columnName!=columnVal:\n            #get data type of compareName column\n            data_type_columnName = get_data_type(columnName)\n\n            #Decide type or correlation!\n            type_corr = \"Pearson\"\n            if data_type_columnVal == \"Categorical\" and data_type_columnName == \"Categorical\":\n                type_corr = \"chi\"\n            elif data_type_columnVal == \"Continuous\" and data_type_columnName == \"Continuous\":\n                type_corr = \"Pearson\"\n            else:\n                type_corr = \"other\"\n        \n            retVal = False\n            if type_corr == \"chi\":\n                num, retVal = chi_squared(table_name, columnVal, columnName)\n            elif type_corr == \"Pearson\":\n                #Should we bother with correlation min?\n                num, retVal = PearsonCorrelation(table_name, columnVal, columnName)\n            else:\n                num, retVal = corrCatToCon(table_name, columnVal, columnName, data_type_columnVal, data_type_columnName)\n            \n            if retVal == \"Strong\":\n                #Retrieve the existing list\n                existing_list = sii_correlations[save_name[i]].get(columnVal)\n                existing_list.append(columnName)\n                sii_correlations[save_name[i]][columnVal] = list(set(existing_list))\n            continue\n        \n        #If columnName is None or columnName == columnVal, recallibrate!\n        for columnCompare in table_name.columns:\n            \n            #no point in comparing to self\n            if columnCompare == columnVal:\n                continue\n                \n            #get data dtype of the column we might add!\n            data_type_columnCompare = get_data_type(columnCompare)\n            \n            #Decide type or correlation!\n            type_corr = \"Pearson\"\n            if data_type_columnVal == \"Categorical\" and data_type_columnCompare == \"Categorical\":\n                type_corr = \"chi\"\n            elif data_type_columnVal == \"Continuous\" and data_type_columnCompare == \"Continuous\":\n                type_corr = \"Pearson\"\n            else:\n                type_corr = \"other\"\n                \n            retVal = False\n            #Is the column correlated?\n            if type_corr == \"chi\":\n                num, retVal = chi_squared(table_name, columnVal, columnCompare)\n            elif type_corr == \"Pearson\":\n                #Should we bother with correlation min?\n                num, retVal = PearsonCorrelation(table_name, columnVal, columnCompare)\n            else:\n                num, retVal = corrCatToCon(table_name, columnVal, columnCompare, data_type_columnVal, data_type_columnCompare)\n                \n            if retVal == \"Strong\":\n                correlations[columnVal].append(columnCompare)\n        \n        #Make sure only unique values\n        correlations[columnVal] = list(set(correlations[columnVal]))\n            \n        #We just want to update the list\n        if columnName == columnVal:\n            sii_correlations[save_name[i]][columnVal] = correlations[columnVal]\n                 \n    return correlations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:47:44.861338Z","iopub.execute_input":"2024-12-03T03:47:44.861769Z","iopub.status.idle":"2024-12-03T03:47:44.874860Z","shell.execute_reply.started":"2024-12-03T03:47:44.861730Z","shell.execute_reply":"2024-12-03T03:47:44.873356Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Save the correlations for EACH column in EACH table within a dictionary - this is to be used as a function so it can be recalculated every time a column has missing values filled in\n    * Correlation will look like: column1:[column4,column6,column8]\n* If we recieve a column_name, it is a sign that we want to calculate the correlation SPECIFICALLY between all columns and THAT column, keeping or removing it accordingly\n    * if the column IS correlated, add it to the list of correlated features! \n    * All while making sure there are no duplicates by turning it into a set and back into a list\n* uses different correlation method for each\n    * Pearson for both continuous\n    * Chi and Cramer for both categorical\n    * Point biseral OR ANOVA for categorical to continuous","metadata":{}},{"cell_type":"code","source":"for i in range (len(tables_to_show)):\n    displayCorr(tables_to_show[i].drop(\"id\", axis = 1))\n    sii_correlations[save_name[i]] = recalibrateCorrelations(tables_to_show[i])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:29:58.928035Z","iopub.execute_input":"2024-12-03T03:29:58.928434Z","iopub.status.idle":"2024-12-03T03:36:30.888109Z","shell.execute_reply.started":"2024-12-03T03:29:58.928397Z","shell.execute_reply":"2024-12-03T03:36:30.887022Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* For each table, callibrate the table correaltions and save\n* Will be saved as :\n    * {none:{column1:[column4,column6,column8], column2:[column3,column6,column7],...}, mild:{column1:[column4,column6,column8], column2:[column3,column6,column7],...},...}","metadata":{}},{"cell_type":"markdown","source":"# Impute Missing values based on correlations!","metadata":{}},{"cell_type":"code","source":"# Train a Ridge regression model (L2-regularized linear regression) because we have a small data set based off of correlations\ndef imputeRegression(X_train, y_train, X_test, target_column, test_name):\n    \n    #First optimize alpha\n    # Define a range of alphas to test for Ridge Regression (log scale or linear scale)\n    alphas = np.linspace(0.01, 10, 100)\n    # Initialize RidgeCV to automatically choose the best alpha using LOOCV \n    ridge_cv = RidgeCV(alphas=alphas, store_cv_values=True)\n    # Fit RidgeCV to the training data\n    ridge_cv.fit(X_train, y_train)\n    # The best alpha chosen by RidgeCV\n    best_alpha = ridge_cv.alpha_\n    \n    ridge = Ridge(alpha=best_alpha)\n    ridge.fit(X_train, y_train)\n    \n    test_name.loc[X_test.index, target_column] = ridge.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:46:45.634233Z","iopub.execute_input":"2024-12-03T03:46:45.634634Z","iopub.status.idle":"2024-12-03T03:46:45.641457Z","shell.execute_reply.started":"2024-12-03T03:46:45.634600Z","shell.execute_reply":"2024-12-03T03:46:45.640363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We chose ridge regression because:\r\n\r\nwe are doing this based off of correlated features\r\nwe want to try and prevent overfitting, which may happen because we have noisy data\r\nwe can use cross validation because since there are not many features, there is risk of underfitting\r\nbasically, uses the X-train and splits it to test on 1, then finds the overall average error for each alpha\r\nThe alpha that has the least error is the optimal alpha to use\r\nWe then use this alpha to create a normal regression model to find the missing target values\r\nfewer features also means less computational expense!","metadata":{}},{"cell_type":"code","source":"def imputeKnn(X_train, y_train, X_test, target_column, test_name):\n    # Scale the features\n    scaler_x= StandardScaler()\n    X_train_scaled = scaler_x.fit_transform(X_train)\n    X_test_scaled = scaler_x.transform(X_test)\n    \n    y_train = np.round(y_train).astype(int)\n    \n    #Find best K value based on accuracy\n    max_k = np.ceil(np.sqrt(X_train.shape[0])).astype(int)\n    k_vals = range(1,max_k+1)\n    \n    # List to store cross-validation results for each K\n    cv_scores = []\n\n    # Evaluate KNN with different K values utilizing LOOCV\n    for k in k_vals:\n        knn = KNeighborsRegressor(n_neighbors=k)\n        scores = cross_val_score(knn, X_train_scaled, y_train, cv=2, scoring='neg_mean_squared_error')  # Use MSE for regression\n        cv_scores.append(np.mean(scores))  # Store mean cross-validation score\n        \n    best_k = k_vals[np.argmax(cv_scores)]\n\n    # Initialize KNN model\n    knn = KNeighborsClassifier(n_neighbors=best_k)\n\n    # Train the model\n    knn.fit(X_train_scaled, y_train)\n    \n    test_name.loc[X_test.index, target_column] = knn.predict(X_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:46:49.268354Z","iopub.execute_input":"2024-12-03T03:46:49.269226Z","iopub.status.idle":"2024-12-03T03:46:49.277135Z","shell.execute_reply.started":"2024-12-03T03:46:49.269181Z","shell.execute_reply":"2024-12-03T03:46:49.275911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"KNN Classification\n* FIRST scale all features! must be done in KNN since we need to normalize for most accurate predictions\n* SECOND find optimal K values\n    * Max K val is sqrt of number of training row data availibe\n    * test for all K increments of one, utilizing LOOCV\n    * np.armin returns index with smallest error \n* Then uses the optimal K to find the Y values","metadata":{}},{"cell_type":"code","source":"def fillFeatures(table_name,i):\n        \n    valuesAdded = 0\n    \n    priorCorrelations = sii_correlations[save_name[i]].values()\n    \n    #Count and sort the missing values within the table\n    missing_counts = table_name.isna().sum()\n    sorted_columns = missing_counts[missing_counts>0].sort_values().index\n\n    #Calculate the missing values starting from those with the LEAST missing values first\n    for target_column in sorted_columns:\n\n        #Get the correlated features for the given target column\n        correlated_features = sii_correlations[save_name[i]].get(target_column)\n\n        #handle situation in which there are no correlated features\n        if not correlated_features or correlated_features is None or len(correlated_features) == 1:\n            continue\n\n        #Separate rows with missing values from non-missing values in the target column\n        df_not_missing = table_name.dropna(subset=[target_column])\n        df_missing = table_name[table_name[target_column].isna()]\n\n        #for training data, replace with mean/median/mode depending on the type of data\n        for t in range(len(correlated_features)):\n            data_type = get_data_type(correlated_features[t])\n            imputeValues(df_not_missing, correlated_features[t], table_name, data_type)\n            #If less than 5% of test data is missing, IMPUTE DATA\n            if(df_missing.shape[0] < table_name.shape[0]*0.05):\n                imputeValues(df_missing, correlated_features[t], table_name, data_type)\n\n        # Define features (X) and target (y) for training\n        X_train = df_not_missing[correlated_features]\n        y_train = df_not_missing[target_column]\n\n        #Handle test data\n        #Drop all rows that have any null data\n        X_test = df_missing[correlated_features].dropna()\n        #If it ends up dropping all rows, see if you can drop columns\n        if (X_test.empty):\n            X_test = df_missing[correlated_features]\n            empty_counts = X_test.isna().sum()\n            #Sort in order of most NaN values\n            to_drop = sorted(empty_counts.items(), key = lambda item: item[1], reverse = True)\n            \n            #Update X_train and X_test accordingly\n            allValues = [column for column, _ in to_drop]\n            numToDrop = 0\n            while X_test.shape[1]>2 and X_test.dropna().empty and numToDrop < len(allValues):\n                X_train = X_train.drop(allValues[numToDrop], axis=1)\n                X_test = X_test.drop(allValues[numToDrop], axis=1)\n                numToDrop+=1\n            X_test = X_test.dropna() \n            #If still empty, not gonna work. just move on\n            if (X_test.empty):\n                continue\n            \n        y_test_rows = X_test.index\n\n        #NOW train the data! If target column is continuous, use regression. If categorical , utilizing knn\n        data_type = get_data_type(target_column)\n        if data_type == \"Categorical\":\n            imputeKnn(X_train, y_train, X_test, target_column, table_name)\n        else:\n            imputeRegression(X_train, y_train, X_test, target_column, table_name)\n            \n        #There were more values added!!\n        valuesAdded +=1\n\n        #NOW recalibrate the correlation values since some missing values have been filled!\n        recalibrateCorrelations(table_name, target_column, i)\n    \n    #If no values were added, there is no purpose in repeating. If there were, maybe we have a chance!\n    if(valuesAdded == 0):\n        return False\n    return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:46:54.781658Z","iopub.execute_input":"2024-12-03T03:46:54.782065Z","iopub.status.idle":"2024-12-03T03:46:54.794502Z","shell.execute_reply.started":"2024-12-03T03:46:54.782030Z","shell.execute_reply":"2024-12-03T03:46:54.793343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* we want to start filling in the missing values BEGINNING from those that are missing the LEAST. After each new column has been filled, re-calibrate correlations\n* For each column, we get the correlated features as found in prior methods\n* Checks:\n    * are there at least 2 correlated features for us to work with?\n    * Is all the train data availible? If not, can we temporarily impute all the missing data\n        * If data<5% missing impute using mean/median/mode!\n            * mean for continuous\n            * mode for categorical\n    * Do we have data for the Test data filled?\n        * If data<5% missing impute using mean/median/mode!\n        * otherwise drop all rows with any missing values\n        * if after dropping rows we have none left, start dropping columns\n        * If still none, move on\n* Calibrate the missing values!\n    * Regression for continuous\n    * KNN for categorical\n* Re-callibrate correlations in relation to specific column since values were added!\n* if no values were added, stop the loop. Otherwise, continue again!\n\n//TODO\n1) Try tree instead of KNN\n2) Encoding categorical data and passing it to Linear Regression or KNN models is not a good practice since these models will perceive these data values as non-discrete","metadata":{}},{"cell_type":"code","source":"#fill features UNTIL most have been filled - we reclaculate correlations and so forth every time there is an update to the table.\n#Impute values that are missing less then 5%\n\nbefore_table = tables_to_show[0].copy()\ntoContinue = True\nwhile(toContinue):\n    toContinue = fillFeatures(tables_to_show[0],0)\nfill_PCIAT(tables_to_show[0])\nlessThen5(tables_to_show[0])\nafter_table = tables_to_show[0]\n\nprint(f\"For the {save_name[0]} sii values, the change is:\")\ndisplayPie(before_table,after_table)\n\n\nbefore_table = tables_to_show[1].copy()\ntoContinue = True\nwhile(toContinue):\n    toContinue = fillFeatures(tables_to_show[1],1)\nfill_PCIAT(tables_to_show[1])\nlessThen5(tables_to_show[1])\nafter_table = tables_to_show[1]\n\nprint(f\"For the {save_name[1]} sii values, the change is:\")\ndisplayPie(before_table,after_table)\n\n\nbefore_table = tables_to_show[2].copy()\ntoContinue = True\nwhile(toContinue):\n    toContinue = fillFeatures(tables_to_show[2],2)\nfill_PCIAT(tables_to_show[2])\nlessThen5(tables_to_show[2])\nafter_table = tables_to_show[2]\n\nprint(f\"For the {save_name[2]} sii values, the change is:\")\ndisplayPie(before_table,after_table)\n\n\nbefore_table = tables_to_show[3].copy()\ntoContinue = True\nwhile(toContinue):\n    toContinue = fillFeatures(tables_to_show[3],3)\nfill_PCIAT(tables_to_show[3])\nafter_table = tables_to_show[3]\nlessThen5(tables_to_show[3])\n\nprint(f\"For the {save_name[3]} sii values, the change is:\")\ndisplayPie(before_table,after_table)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:47:53.617081Z","iopub.execute_input":"2024-12-03T03:47:53.617472Z","iopub.status.idle":"2024-12-03T03:54:26.557778Z","shell.execute_reply.started":"2024-12-03T03:47:53.617438Z","shell.execute_reply":"2024-12-03T03:54:26.556786Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# COMBINE all and re-do testing!","metadata":{}},{"cell_type":"code","source":"#Show all the features and then put back into training file!\nprint(\"Individual Feature Updates:\")\nfilledFeaturesPlot()\ntrain.loc[tables_to_show[0].index] = tables_to_show[0]\ntrain.loc[tables_to_show[1].index] = tables_to_show[1]\ntrain.loc[tables_to_show[2].index] = tables_to_show[2]\ntrain.loc[tables_to_show[3].index] = tables_to_show[3]\n\n#make sure to still fill values and understand accuracy measures:\nfill_PCIAT(train)\nclusterEntropy(train, method = \"Cosine\")\nlessThen5(train)\nfitness_ranges(Fitness_table)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Now find and fill correlations based on sii and fill accordingly!\nsii_correlations[\"train\"] = recalibrateCorrelations(train)\ntoContinue = True\nwhile toContinue:\n    toContinue = fillFeatures(train,4)\n    \n#Fill data again for values:\nfill_PCIAT(train)\nlessThen5(train)\n\n#Show difference before dropping data:\nprint(\"Total Difference: \")\ndisplayPie(train_before, train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Drop training data with NaN as a sii as well as  >50% of row data missing\nfiltered_df = train[train['sii'].isna()]\nminSize = (filtered_df.shape[1]-1)/2\nsumsOfNon = filtered_df.isna().sum(axis = 1)\ndrop_indexes = [theIndex[0] for theIndex in sumsOfNon.items() if theIndex[1]>=minSize]\ntrain.drop(drop_indexes, inplace = True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Training Data Updates:\")\nfilledIndividualPlot(train)\nprint(\"Difference in Age metrics:\")\nbar_graph(train_before[\"Basic_DemosAge\"].value_counts(), \"Age distribution Before\")\nbar_graph(train[\"Basic_DemosAge\"].value_counts(), \"Age distribution After\")\nprint(\"Difference in Sex metrics:\")\nbar_graph(train_before[\"Basic_DemosSex\"].value_counts(), \"Sex distribution Before\")\nbar_graph(train[\"Basic_DemosSex\"].value_counts(), \"Sex distribution After\")\n#Show sii distribution\nbar_graph(train_before[\"sii\"].value_counts(), \"Sii Distribution Before\")\nbar_graph(train[\"sii\"].value_counts(), \"Sii Distribution After\")\n\nprint(\"Total Difference: \")\ndisplayPie(train_before, train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"displayCorr(train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DO SOME TESTING ON TRAINING SET VIA MODELS","metadata":{}},{"cell_type":"code","source":"from sklearn.datasets import make_classification\nfrom imblearn.over_sampling import SMOTENC\n\ndef SMOTEtest(X_train, Y_train, categorical_features):\n\n    smote_nc = SMOTENC(categorical_features=categorical_features, random_state=42)\n    X_resampled, y_resampled = smote_nc.fit_resample(X, y)\n    \n    return X_resampled, y_resampled","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"SMOTE testing!","metadata":{}},{"cell_type":"code","source":"# Split the dataset into training and testing sets\nX = train.iloc[:, :-1]\ny = train.iloc[:, -1]\n\nx_train, x_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\ncategorical_features = []\nfor i, columns in enumerate(x_train.columns):\n    if get_data_type(columns) == \"Categorical\"\n        categorical_features.append(i)\n\nnew_X, new_Y = SMOTEtest(x_train, y_train, categorical_features)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_auc_score, RocCurveDisplay\nfrom sklearn.model_selection import cross_val_score, RepeatedStratifiedKFold\nfrom sklearn.metrics import classification_report, accuracy_score, precision_recall_curve\nimport numpy as np \n\ndef model(classifier, X_train, Y_train, X_test, Y_test):\n    \"\"\"\n    Function to train a model, evaluate performance, and display metrics\n    \"\"\"\n    # Train the bagging classifier\n    classifier.fit(X_train, Y_train)\n    \n    # Predict on the test set\n    prediction = classifier.predict(X_test)\n    \n    # create cross-validation rules \n    cv = RepeatedStratifiedKFold(n_splits=10, n_repeats=3, random_state=1)\n    \n    # Print metrics\n    print(\"Accuracy: \", '{0:.2%}'.format(accuracy_score(Y_test, prediction)))\n    # Perfrom CV\n    print(\"Cross-Validation Score (ROC_AUC): \", \n          '{0:.2%}'.format(cross_val_score(classifier, X_train, Y_train, cv=cv, scoring='roc_auc').mean()))\n    print(\"ROC_AUC Score: \", '{0:.2%}'.format(roc_auc_score(Y_test, classifier.predict_proba(X_test)[:, 1])))\n    # Plot ROC Curve\n    RocCurveDisplay.from_estimator(classifier, X_test, Y_test)\n    plt.title('ROC AUC Plot')\n    plt.show()\n\ndef model_evaluation(classifier, X_train, Y_train, X_test, Y_test):\n    \"\"\"\n    Function to evaluate the classifier using confusion matrix and classification report\n    \"\"\"\n    # Confusion Matrix\n    cm = confusion_matrix(Y_test, classifier.predict(X_test))\n    names = ['True Neg', 'False Pos', 'False Neg', 'True Pos']\n    counts = [value for value in cm.flatten()]\n    percentages = ['{0:.2%}'.format(value) for value in cm.flatten() / np.sum(cm)]\n    labels = [f'{v1}\\n{v2}\\n{v3}' for v1, v2, v3 in zip(names, counts, percentages)]\n    labels = np.asarray(labels).reshape(2, 2)\n    \n    # Plot Confusion Matrix\n    sns.heatmap(cm, annot=labels, cmap='Blues', fmt='', cbar=False)\n    plt.title('Confusion Matrix')\n    plt.show()\n    \n    # Classification Report\n    print(\"\\nClassification Report:\\n\")\n    print(classification_report(Y_test, classifier.predict(X_test)))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import BaggingClassifier\n\n# lr classifier\nclassifierLR = LogisticRegression(random_state=42, solver='lbfgs', max_iter=1500)\n\n# Bagging Classifier\nbaggingLR = BaggingClassifier(\n    estimator=classifierLR,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n# Train and evaluate the model\nprint(\"No Bagging without SMOTE:\")\nmodel(classifierLR, x_train, y_train, x_test, y_test)\nmodel_evaluation(classifierLR, x_train, y_train, x_test, y_test)\n\nprint(\"Bagging without SMOTE:\")\nmodel(baggingLR, x_train, y_train, x_test, y_test)\nmodel_evaluation(baggingLR, x_train, y_train, x_test, y_test)\n\n#NOW utilize SMOTE to test\nprint(\"No Bagging with SMOTE:\")\nmodel(classifierLR, new_X, new_Y, x_test, y_test)\nmodel_evaluation(classifierLR, new_X, new_Y, x_test, y_test)\n\nprint(\"Bagging with SMOTE:\")\nmodel(baggingLR, new_X, new_Y, x_test, y_test)\nmodel_evaluation(baggingLR, new_X, new_Y, x_test, y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nclassifierKNN = KNeighborsClassifier(n_neighbors=11)\n\n# Bagging Classifier\nbaggingKNN = BaggingClassifier(\n    estimator=classifierKNN,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\nclassifier_SMOTEKNN = KNeighborsClassifier(n_neighbors=7)\n\n# Bagging Classifier SMOTE\nbagging_SMOTE = BaggingClassifier(\n    estimator=classifier_SMOTEKNN,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n# Train and evaluate the model\nprint(\"No Bagging without SMOTE:\")\nmodel(classifierKNN, x_train, y_train, x_test, y_test)\nmodel_evaluation(classifierKNN, x_train, y_train, x_test, y_test)\n\nprint(\"Bagging without SMOTE:\")\nmodel(baggingKNN, x_train, y_train, x_test, y_test)\nmodel_evaluation(baggingKNN, x_train, y_train, x_test, y_test)\n\n#NOW utilize SMOTE to test\nprint(\"No Bagging with SMOTE:\")\nmodel(classifier_SMOTEKNN, new_X, new_Y, x_test, y_test)\nmodel_evaluation(classifier_SMOTEKNN, new_X, new_Y, x_test, y_test)\n\nprint(\"Bagging with SMOTE:\")\nmodel(bagging_SMOTEKNN, new_X, new_Y, x_test, y_test)\nmodel_evaluation(bagging_SMOTEKNN, new_X, new_Y, x_test, y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import cross_val_score\nimport matplotlib.pyplot as plt\n\n# Range of n_neighbors to test\nn_values = range(1, 31)\nmean_accuracies = []\n\n# Loop over n_values\nfor n in n_values:\n    knn = KNeighborsClassifier(n_neighbors=n)\n    scores = cross_val_score(knn, x_train, y_train, cv=10, scoring='accuracy')  # 5-fold cross-validation\n    mean_accuracies.append(scores.mean())\n\n# Plot the results\nplt.figure(figsize=(10, 6))\nplt.plot(n_values, mean_accuracies, marker='o')\nplt.title('Finding the Best n_neighbors for KNN', fontsize=14, fontweight='bold')\nplt.xlabel('Number of Neighbors (n)', fontsize=12)\nplt.ylabel('Mean Accuracy (Cross-Validation)', fontsize=12)\nplt.grid()\nplt.show()\n\nmean_accuracies_SMOTE = []\n# Loop over n_values\nfor n in n_values:\n    knn = KNeighborsClassifier(n_neighbors=n)\n    scores = cross_val_score(knn, new_X, new_Y, cv=10, scoring='accuracy')  # 5-fold cross-validation\n    mean_accuracies_SMOTE.append(scores.mean())\n\n# Plot the results\nplt.figure(figsize=(10, 6))\nplt.plot(n_values, mean_accuracies_SMOTE, marker='o')\nplt.title('Finding the Best n_neighbors for KNN', fontsize=14, fontweight='bold')\nplt.xlabel('Number of Neighbors (n)', fontsize=12)\nplt.ylabel('Mean Accuracy (Cross-Validation)', fontsize=12)\nplt.grid()\nplt.show()\n\n# Find the best n\nbest_n = n_values[mean_accuracies.index(max(mean_accuracies))]\nbest_n_SMOTE = n_values[mean_accuracies_SMOTE.index(max(mean_accuracies_SMOTE))]\nprint(f\"The best value for n_neighbors no SMOTE is {best_n} with a cross-validated accuracy of {max(mean_accuracies):.2%}\")\nprint(f\"The best value for n_neighbors with SMOTE is {best_n_SMOTE} with a cross-validated accuracy of {max(mean_accuracies):.2%}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Perform Grid Search to find the best parameters\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV\n\nparam_grid = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4]\n}\n\n# Initialize Random Forest Classifier\nrf = RandomForestClassifier(random_state=1)\n\n\n# Perform Grid Search\ngrid_search = GridSearchCV(rf, param_grid, cv=5, scoring='accuracy')\ngrid_search.fit(x_train, y_train)\n\n# Print the best parameters\nbest_params = grid_search.best_params_\nprint(\"Best Parameters:\", best_params)\n\n# Step 2: Train the model with the best parameters\nbest_rf = RandomForestClassifier(**best_params, random_state=1)\n\n# Bagging Classifier\nbaggingRF = BaggingClassifier(\n    estimator=best_rf,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n\n#now for SMOTE\n# Perform Grid Search\ngrid_search_SMOTE = GridSearchCV(rf, param_grid, cv=5, scoring='accuracy')\ngrid_search_SMOTE.fit(new_X, new_Y)\n\n# Print the best parameters\nbest_params_SMOTE = grid_search_SMOTE.best_params_\nprint(\"Best Parameters for SMOTE:\", best_params_SMOTE)\n\n# Step 2: Train the model with the best parameters\nbest_rf_SMOTE = RandomForestClassifier(**best_params_SMOTE, random_state=1)\n\n# Bagging Classifier SMOTE\nbagging_SMOTERF = BaggingClassifier(\n    estimator=best_rf_SMOTE,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n# Train and evaluate the model\nprint(\"No Bagging without SMOTE:\")\nmodel(best_rf, x_train, y_train, x_test, y_test)\nmodel_evaluation(best_rf, x_train, y_train, x_test, y_test)\n\nprint(\"Bagging without SMOTE:\")\nmodel(baggingRF, x_train, y_train, x_test, y_test)\nmodel_evaluation(baggingRF, x_train, y_train, x_test, y_test)\n\n#NOW utilize SMOTE to test\nprint(\"No Bagging with SMOTE:\")\nmodel(best_rf_SMOTE, new_X, new_Y, x_test, y_test)\nmodel_evaluation(best_rf_SMOTE, new_X, new_Y, x_test, y_test)\n\nprint(\"Bagging with SMOTE:\")\nmodel(bagging_SMOTERF, new_X, new_Y, x_test, y_test)\nmodel_evaluation(bagging_SMOTERF, new_X, new_Y, x_test, y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nclassifierXG = XGBClassifier(random_state=1, eval_metric='logloss')\n\n# Bagging Classifier\nbaggingXG = BaggingClassifier(\n    estimator=classifierXG,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n# Train and evaluate the model\nprint(\"No Bagging without SMOTE:\")\nmodel(classifierXG, x_train, y_train, x_test, y_test)\nmodel_evaluation(classifierXG, x_train, y_train, x_test, y_test)\n\nprint(\"Bagging without SMOTE:\")\nmodel(baggingXG, x_train, y_train, x_test, y_test)\nmodel_evaluation(baggingXG, x_train, y_train, x_test, y_test)\n\n#NOW utilize SMOTE to test\nprint(\"No Bagging with SMOTE:\")\nmodel(classifierXG, new_X, new_Y, x_test, y_test)\nmodel_evaluation(classifierXG, new_X, new_Y, x_test, y_test)\n\nprint(\"Bagging with SMOTE:\")\nmodel(baggingXG, new_X, new_Y, x_test, y_test)\nmodel_evaluation(baggingXG, new_X, new_Y, x_test, y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\nclassifierNB=GaussianNB()\n\n# Bagging Classifier\nbaggingNB = BaggingClassifier(\n    estimator=classifierNB,\n    n_estimators=10,  # Number of bootstrap iterations\n    max_samples=0.8,  # Fraction of data sampled in each iteration\n    random_state=42\n)\n\n# Train and evaluate the model\nprint(\"No Bagging without SMOTE:\")\nmodel(classifierNB, x_train, y_train, x_test, y_test)\nmodel_evaluation(classifierNB, x_train, y_train, x_test, y_test)\n\nprint(\"Bagging without SMOTE:\")\nmodel(baggingNB, x_train, y_train, x_test, y_test)\nmodel_evaluation(baggingNB, x_train, y_train, x_test, y_test)\n\n#NOW utilize SMOTE to test\nprint(\"No Bagging with SMOTE:\")\nmodel(classifierNB, new_X, new_Y, x_test, y_test)\nmodel_evaluation(classifierNB, new_X, new_Y, x_test, y_test)\n\nprint(\"Bagging with SMOTE:\")\nmodel(baggingNB, new_X, new_Y, x_test, y_test)\nmodel_evaluation(baggingNB, new_X, new_Y, x_test, y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}