{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import and general settings\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n# settings so that pandas shows all columns and rows, because per default, it \n# shows only some if the dataset is large\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:26.672044Z","iopub.execute_input":"2025-01-11T15:08:26.672432Z","iopub.status.idle":"2025-01-11T15:08:27.924220Z","shell.execute_reply.started":"2025-01-11T15:08:26.672394Z","shell.execute_reply":"2025-01-11T15:08:27.923187Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"","metadata":{"execution":{"iopub.status.busy":"2025-01-10T21:30:59.883919Z","iopub.execute_input":"2025-01-10T21:30:59.884416Z","iopub.status.idle":"2025-01-10T21:30:59.963102Z","shell.execute_reply.started":"2025-01-10T21:30:59.884385Z","shell.execute_reply":"2025-01-10T21:30:59.961809Z"}}},{"cell_type":"code","source":"df = pd.read_csv('../input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('../input/child-mind-institute-problematic-internet-use/test.csv')\n\n# drop the PCAT-columns from the dataset, because the target value (sii) is calculated\n# with them and in the test dataset, we won't have them to predict sii\n\ncolumns_to_drop = [\n    'PCIAT-Season',\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', \n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', \n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', \n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', \n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total'\n]\ndf = df.drop(columns=columns_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:27.925233Z","iopub.execute_input":"2025-01-11T15:08:27.925744Z","iopub.status.idle":"2025-01-11T15:08:28.017938Z","shell.execute_reply.started":"2025-01-11T15:08:27.925710Z","shell.execute_reply":"2025-01-11T15:08:28.017005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# merge the columns PAQ_A-PAQ_A_Total and PAQ_C-PAQ_C_Total into one column called PAQ_Total\ndef calculate_paq_total(row):\n    a = row['PAQ_A-PAQ_A_Total']\n    c = row['PAQ_C-PAQ_C_Total']\n    \n    if pd.notnull(a) and pd.notnull(c): \n        return (a + c) / 2 # if both values are there, calculate average\n    elif pd.notnull(a):\n        return a\n    elif pd.notnull(c):\n        return c\n    else: \n        return None\n\n# create a new column and fill it with help of the created function\ndf['PAQ_Total'] = df.apply(calculate_paq_total, axis=1)\n# drop the old two columns\ndf = df.drop(columns=['PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.019181Z","iopub.execute_input":"2025-01-11T15:08:28.019578Z","iopub.status.idle":"2025-01-11T15:08:28.084598Z","shell.execute_reply.started":"2025-01-11T15:08:28.019530Z","shell.execute_reply":"2025-01-11T15:08:28.083653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# merge the columns Fitness_Endurance-Time_Mins and Fitness_Endurance-Time_Sec, because\n# seconds alone have no meaning\n\n# create new column: Fitness_Endurance-Time\ndf['Fitness_Endurance-Time'] = df['Fitness_Endurance-Time_Mins'] + (df['Fitness_Endurance-Time_Sec'] / 60)\n\n# drop old columns\ndf = df.drop(columns=['Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'BIA-Season', 'Fitness_Endurance-Season', 'PAQ_A-Season'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.087101Z","iopub.execute_input":"2025-01-11T15:08:28.087434Z","iopub.status.idle":"2025-01-11T15:08:28.094733Z","shell.execute_reply.started":"2025-01-11T15:08:28.087408Z","shell.execute_reply":"2025-01-11T15:08:28.093746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Different types of data\n* numerical discrete data: **numerical_discrete_columns**\n* numerical continuous data: **numerical_continuous_columns**\n* categorical data (but expressed in numbers): **categorical_columns_expressedInNumbers**\n* categorical data (expressed in string): **categorical_columns_expressedInText**\n* ordinal data","metadata":{}},{"cell_type":"code","source":"categorical_columns_expressedInNumbers = ['Basic_Demos-Sex', 'PreInt_EduHx-computerinternet_hoursday', 'FGC-FGC_TL_Zone', 'FGC-FGC_CU_Zone', \n                               'FGC-FGC_PU_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone', 'BIA-BIA_Activity_Level_num', \n                              'BIA-BIA_Frame_num', 'FGC-FGC_GSD_Zone','FGC-FGC_GSND_Zone']\n\n# Konvertiere die Spalten in den Typ 'category'\nfor col in categorical_columns_expressedInNumbers:\n    df[col] = df[col].astype('category')\n\ncategorical_columns_expressedInText = df.select_dtypes(include=\"object\").columns\n\nnumerical_discrete_columns = ['Basic_Demos-Age', 'CGAS-CGAS_Score', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                     'Fitness_Endurance-Max_Stage', 'FGC-FGC_CU', 'FGC-FGC_PU',\n                                'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T']\nnumerical_continuous_columns = ['Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                       'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL',\n                       'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW',\n                       'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM',\n                       'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'Fitness_Endurance-Time',\n                       'PAQ_Total']\n\ncategorical_columns_expressedInText = list(categorical_columns_expressedInText)\ncategorical_columns_expressedInNumbers = list(categorical_columns_expressedInNumbers)\n\nif \"id\" in categorical_columns_expressedInText:\n    categorical_columns_expressedInText.remove(\"id\")\n\n# Combine categorical columns\nall_categorical_columns = categorical_columns_expressedInNumbers + categorical_columns_expressedInText\n\nnumerical_discrete_columns = list(numerical_discrete_columns)\nnumerical_continuous_columns = list(numerical_continuous_columns)\n\nall_numerical_columns = numerical_discrete_columns + numerical_continuous_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.096333Z","iopub.execute_input":"2025-01-11T15:08:28.096641Z","iopub.status.idle":"2025-01-11T15:08:28.126254Z","shell.execute_reply.started":"2025-01-11T15:08:28.096615Z","shell.execute_reply":"2025-01-11T15:08:28.125172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Negative value treatment","metadata":{}},{"cell_type":"code","source":"def whisker(col):\n    q1,q3=np.percentile(col, [25, 75])\n    iqr=q3-q1\n    lw=q1-1.5*iqr\n    uw=q3+1.5*iqr\n    return lw, uw\n\n\ndef replace_negatives_with_median(df, columns):\n    for col in columns:\n        median_value = df[col].median() \n        df[col] = np.where(df[col] <= 0, median_value, df[col])\n    return df\n\n\n# def replace_negatives_with_lower_whisker(df, columns):\n#     for col in columns:\n#         lw = whisker(df[col])[0] \n#         df[col] = np.where(df[col] <= 0, lw, df[col])\n#     return df\n\n\ncolumns_with_negative_values = ['Physical-BMI', 'Physical-Weight', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                               'BIA-BIA_BMR', 'BIA-BIA_FMI', 'BIA-BIA_Fat']\ndf = replace_negatives_with_median(df, columns_with_negative_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.127452Z","iopub.execute_input":"2025-01-11T15:08:28.127845Z","iopub.status.idle":"2025-01-11T15:08:28.148318Z","shell.execute_reply.started":"2025-01-11T15:08:28.127803Z","shell.execute_reply":"2025-01-11T15:08:28.147324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Drop rows that make no sense","metadata":{}},{"cell_type":"code","source":"def dropRows(df):\n    df = df.drop(df[df['BIA-BIA_FFM'] > 6000].index)\n    df = df.drop(df[df['BIA-BIA_BMC'] > 100].index)\n    #reset index if you want to remain it consecutive\n    df = df.reset_index(drop=True)\n    return df\n\ndf = dropRows(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.149204Z","iopub.execute_input":"2025-01-11T15:08:28.149498Z","iopub.status.idle":"2025-01-11T15:08:28.167140Z","shell.execute_reply.started":"2025-01-11T15:08:28.149460Z","shell.execute_reply":"2025-01-11T15:08:28.166091Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Missing value treatment","metadata":{}},{"cell_type":"code","source":"# Missing Value Treatment Attempt 1: Impute with median and mode\ndf_i_w_m_a_m = df.copy()\n# Für numerische Spalten\nfor i in all_numerical_columns:\n    df_i_w_m_a_m[i] = df_i_w_m_a_m[i].fillna(df_i_w_m_a_m[i].median()) \n\n# Für kategorische Spalten, Werte mit dem Modus auffüllen\nfor i in all_categorical_columns:\n    df_i_w_m_a_m[i] = df_i_w_m_a_m[i].fillna(df_i_w_m_a_m[i].mode()[0])\n\n# Check if values were filled\ndf_i_w_m_a_m.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.168092Z","iopub.execute_input":"2025-01-11T15:08:28.168409Z","iopub.status.idle":"2025-01-11T15:08:28.242239Z","shell.execute_reply.started":"2025-01-11T15:08:28.168384Z","shell.execute_reply":"2025-01-11T15:08:28.241156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Outlier treatment\n- no outlier treatment","metadata":{}},{"cell_type":"markdown","source":"# One-Hot Encoding","metadata":{}},{"cell_type":"code","source":"df_i_w_m_a_m = pd.get_dummies(df_i_w_m_a_m, columns=['Basic_Demos-Sex', 'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'FGC-Season','PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'], drop_first=True)\ndf_i_w_m_a_m.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.243185Z","iopub.execute_input":"2025-01-11T15:08:28.243466Z","iopub.status.idle":"2025-01-11T15:08:28.316792Z","shell.execute_reply.started":"2025-01-11T15:08:28.243441Z","shell.execute_reply":"2025-01-11T15:08:28.315811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Semi-supervised learning","metadata":{}},{"cell_type":"code","source":"\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.semi_supervised import SelfTrainingClassifier\nfrom sklearn.model_selection import train_test_split\n\n# Gelabelte Daten (sii ist bekannt)\nlabeled_data = df_i_w_m_a_m[df_i_w_m_a_m['sii'].notna()]\n\n# Unlabelte Daten (sii ist unbekannt)\nunlabeled_data = df_i_w_m_a_m[df_i_w_m_a_m['sii'].isna()]\n\n# Gelabelte Daten\nids_labeled = labeled_data['id']  # IDs der gelabelten Daten\nX_labeled = labeled_data.drop(['sii', 'id'], axis=1)  # Features ohne 'id' und 'sii'\ny_labeled = labeled_data['sii']  # Zielvariable\n\n\n\n# Ungelabelte Daten\nids_unlabeled = unlabeled_data['id']  # IDs der ungelabelten Daten\nX_unlabeled = unlabeled_data.drop(['sii', 'id'], axis=1)  # Features ohne 'id'\ny_unlabeled = pd.Series([-1] * len(X_unlabeled), index=X_unlabeled.index)  # Zielvariable als -1\n\nX_labeled = X_labeled.reset_index(drop=True)\ny_labeled = y_labeled.reset_index(drop=True)\n\n# Aufteilen der gelabelten Daten in Training und Test\nX_train, X_test, y_train, y_test = train_test_split(\n    X_labeled, y_labeled, test_size=0.2, random_state=42, shuffle=True\n)\n\n# Kombinieren der gelabelten und ungelabelten Daten\nX_combined = pd.concat([X_train, X_unlabeled])\ny_combined = pd.concat([y_train, y_unlabeled])\n\nprint(\"Gibt es Überschneidungen zwischen Training und Test?\")\nprint(X_train.index.intersection(X_test.index))  # Sollte leer sein: Index([])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.317695Z","iopub.execute_input":"2025-01-11T15:08:28.317998Z","iopub.status.idle":"2025-01-11T15:08:28.904462Z","shell.execute_reply.started":"2025-01-11T15:08:28.317963Z","shell.execute_reply":"2025-01-11T15:08:28.903277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.ensemble import BalancedRandomForestClassifier\n\n# Basis-Klassifikator mit BalancedRandomForestClassifier\nbase_model = BalancedRandomForestClassifier(random_state=42)\n\n# SelfTrainingClassifier\nself_training_model = SelfTrainingClassifier(base_model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:28.905536Z","iopub.execute_input":"2025-01-11T15:08:28.905888Z","iopub.status.idle":"2025-01-11T15:08:29.109675Z","shell.execute_reply.started":"2025-01-11T15:08:28.905858Z","shell.execute_reply":"2025-01-11T15:08:29.108326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training des Modells\nself_training_model.fit(X_combined, y_combined)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:29.111131Z","iopub.execute_input":"2025-01-11T15:08:29.111746Z","iopub.status.idle":"2025-01-11T15:08:31.189275Z","shell.execute_reply.started":"2025-01-11T15:08:29.111703Z","shell.execute_reply":"2025-01-11T15:08:31.188231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialisiere eine Variable, um die Anzahl der neuen hochsicheren Labels zu verfolgen\nnew_high_conf_count = 1  # Startwert > 0, um die Schleife zu starten\n\n# Schleife: Wiederholen, solange neue hochsichere Labels gefunden werden\nwhile new_high_conf_count > 0:\n    # Vorhersagen und Wahrscheinlichkeiten für ungelabelte Daten\n    predictions = self_training_model.predict(X_unlabeled)\n    probabilities = self_training_model.predict_proba(X_unlabeled)\n\n    # Sicherheitsschwelle festlegen (z. B. 0.9)\n    confidence_threshold = 0.9\n    high_confidence_indices = (probabilities.max(axis=1) >= confidence_threshold)\n\n    # Hochsichere Vorhersagen und ihre Features extrahieren\n    X_high_conf = X_unlabeled[high_confidence_indices]\n    y_high_conf = predictions[high_confidence_indices]\n\n    # Zähle die Anzahl der neuen hochsicheren Labels\n    new_high_conf_count = len(X_high_conf)\n    print(f\"Neue hochsichere Vorhersagen in dieser Iteration: {new_high_conf_count}\")\n    print(f\"Anzahl der nicht hoch sicheren Vorhersagen: {len(probabilities)-new_high_conf_count}\")\n\n    # Wenn keine neuen hochsicheren Labels mehr vorhanden sind, abbrechen\n    if new_high_conf_count == 0:\n        break\n\n    # Hochsichere Daten zu gelabelten Daten hinzufügen\n    X_labeled = pd.concat([X_train, X_high_conf])\n    y_labeled = pd.concat([y_train, pd.Series(y_high_conf, index=X_high_conf.index)])\n\n    # Entfernen der hochsicheren Daten aus den ungelabelten Daten\n    X_unlabeled = X_unlabeled.drop(index=X_high_conf.index)\n\n    # Zielvariable für verbleibende ungelabelte Daten aktualisieren (-1 bleibt für diese erhalten)\n    y_unlabeled = pd.Series([-1] * len(X_unlabeled), index=X_unlabeled.index)\n\n    # Kombinieren der gelabelten und verbleibenden ungelabelten Daten\n    X_combined = pd.concat([X_labeled, X_unlabeled])\n    y_combined = pd.concat([y_labeled, y_unlabeled])\n\n    # Modell erneut trainieren\n    self_training_model.fit(X_combined, y_combined)\n\nprint(\"Training abgeschlossen. Keine weiteren hochsicheren Vorhersagen verfügbar.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:31.192068Z","iopub.execute_input":"2025-01-11T15:08:31.192361Z","iopub.status.idle":"2025-01-11T15:08:34.429665Z","shell.execute_reply.started":"2025-01-11T15:08:31.192337Z","shell.execute_reply":"2025-01-11T15:08:34.425651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, cohen_kappa_score\n\n# Vorhersagen auf den Testdaten\ny_pred = self_training_model.predict(X_test)\n\n# Confusion Matrix\nconf_matrix = confusion_matrix(y_test, y_pred)\nprint(\"\\nConfusion Matrix:\")\nprint(conf_matrix)\n\n# Classification Report\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))\n\n# Quadratisch gewichtete Kappa-Metrik\nquadratic_kappa = cohen_kappa_score(y_test, y_pred, weights=\"quadratic\")\nprint(f\"\\nQuadratisch gewichtete Kappa-Metrik: {quadratic_kappa:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.432080Z","iopub.execute_input":"2025-01-11T15:08:34.432459Z","iopub.status.idle":"2025-01-11T15:08:34.473450Z","shell.execute_reply.started":"2025-01-11T15:08:34.432426Z","shell.execute_reply":"2025-01-11T15:08:34.471972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Vorhersagen und Wahrscheinlichkeiten für ungelabelte Daten\n# predictions = self_training_model.predict(X_unlabeled)\n# probabilities = self_training_model.predict_proba(X_unlabeled)\n\n# # Sicherheitsschwelle festlegen (z. B. 0.9)\n# confidence_threshold = 0.9\n# high_confidence_indices = (probabilities.max(axis=1) >= confidence_threshold)\n# num_high_confidence = high_confidence_indices.sum()\n# print(f\"Anzahl der hochsicheren Vorhersagen: {num_high_confidence}\")\n# print(f\"Anzahl der nicht hoch sicheren Vorhersagen: {len(probabilities)}\")\n\n# # Hochsichere Vorhersagen und ihre Features extrahieren\n# X_high_conf = X_unlabeled[high_confidence_indices]\n# y_high_conf = predictions[high_confidence_indices]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.474807Z","iopub.execute_input":"2025-01-11T15:08:34.475221Z","iopub.status.idle":"2025-01-11T15:08:34.479947Z","shell.execute_reply.started":"2025-01-11T15:08:34.475157Z","shell.execute_reply":"2025-01-11T15:08:34.478899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Hochsichere Daten zu gelabelten Daten hinzufügen\n# X_labeled = pd.concat([X_labeled, X_high_conf])\n# y_labeled = pd.concat([y_labeled, pd.Series(y_high_conf, index=X_high_conf.index)])\n\n# # Entfernen der hochsicheren Daten aus den ungelabelten Daten\n# X_unlabeled = X_unlabeled.drop(index=X_high_conf.index)\n\n# # Zielvariable für verbleibende ungelabelte Daten aktualisieren (-1 bleibt für diese erhalten)\n# y_unlabeled = pd.Series([-1] * len(X_unlabeled), index=X_unlabeled.index)\n\n# # Kombinieren der gelabelten und verbleibenden ungelabelten Daten\n# X_combined = pd.concat([X_labeled, X_unlabeled])\n# y_combined = pd.concat([y_labeled, y_unlabeled])\n\n# # Modell erneut trainieren\n# self_training_model.fit(X_combined, y_combined)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.480967Z","iopub.execute_input":"2025-01-11T15:08:34.481350Z","iopub.status.idle":"2025-01-11T15:08:34.498976Z","shell.execute_reply.started":"2025-01-11T15:08:34.481317Z","shell.execute_reply":"2025-01-11T15:08:34.497773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# IDs der Testdaten extrahieren\nids_test = test_data['id']\n\n# Testdaten vorbereiten (entfernen der 'id'-Spalte)\nX_test = test_data.drop(['id'], axis=1)\n\n# do necessary transformations, e.g. one-hot encoding\n\nX_test['PAQ_Total'] = X_test.apply(calculate_paq_total, axis=1)\n# drop the old two columns\nX_test = X_test.drop(columns=['PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total'])\n\nX_test['Fitness_Endurance-Time'] = X_test['Fitness_Endurance-Time_Mins'] + (X_test['Fitness_Endurance-Time_Sec'] / 60)\n\n# drop old columns\n\nX_test = X_test.drop(columns=['Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'BIA-Season', 'Fitness_Endurance-Season', 'PAQ_A-Season'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.499967Z","iopub.execute_input":"2025-01-11T15:08:34.500435Z","iopub.status.idle":"2025-01-11T15:08:34.521318Z","shell.execute_reply.started":"2025-01-11T15:08:34.500400Z","shell.execute_reply":"2025-01-11T15:08:34.520291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Missing Value Treatment Attempt 1: Impute with median and mode\nX_test = X_test.copy()\n# Für numerische Spalten\nfor i in all_numerical_columns:\n    X_test[i] = X_test[i].fillna(X_test[i].median()) \n\n# Für kategorische Spalten, Werte mit dem Modus auffüllen\nfor i in all_categorical_columns:\n    X_test[i] = X_test[i].fillna(X_test[i].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.522312Z","iopub.execute_input":"2025-01-11T15:08:34.522672Z","iopub.status.idle":"2025-01-11T15:08:34.573888Z","shell.execute_reply.started":"2025-01-11T15:08:34.522639Z","shell.execute_reply":"2025-01-11T15:08:34.572726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = pd.get_dummies(X_test, columns=['Basic_Demos-Sex', 'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'FGC-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'], drop_first=True)\nX_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.575153Z","iopub.execute_input":"2025-01-11T15:08:34.575563Z","iopub.status.idle":"2025-01-11T15:08:34.660125Z","shell.execute_reply.started":"2025-01-11T15:08:34.575527Z","shell.execute_reply":"2025-01-11T15:08:34.658863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vorhersagen für die Testdaten\nfinal_predictions = self_training_model.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.661325Z","iopub.execute_input":"2025-01-11T15:08:34.661724Z","iopub.status.idle":"2025-01-11T15:08:34.680451Z","shell.execute_reply.started":"2025-01-11T15:08:34.661690Z","shell.execute_reply":"2025-01-11T15:08:34.679255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ergebnisse in einem DataFrame speichern\nsubmission = pd.DataFrame({\n    'id': ids_test,  # IDs der Testdaten\n    'sii': final_predictions  # Vorhergesagte sii-Werte\n})\n\n# Konvertieren der sii-Werte in ganze Zahlen\nsubmission['sii'] = submission['sii'].astype(int)\n\n# Ergebnisse als CSV speichern\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.681643Z","iopub.execute_input":"2025-01-11T15:08:34.682036Z","iopub.status.idle":"2025-01-11T15:08:34.692726Z","shell.execute_reply.started":"2025-01-11T15:08:34.681989Z","shell.execute_reply":"2025-01-11T15:08:34.691438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Absoluten Pfad der Datei anzeigen\nprint(os.path.abspath('submission.csv'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T15:08:34.693686Z","iopub.execute_input":"2025-01-11T15:08:34.694033Z","iopub.status.idle":"2025-01-11T15:08:34.702453Z","shell.execute_reply.started":"2025-01-11T15:08:34.693990Z","shell.execute_reply":"2025-01-11T15:08:34.701395Z"}},"outputs":[],"execution_count":null}]}