{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-18T07:03:08.910539Z","iopub.execute_input":"2024-05-18T07:03:08.910989Z","iopub.status.idle":"2024-05-18T07:03:12.118575Z","shell.execute_reply.started":"2024-05-18T07:03:08.910955Z","shell.execute_reply":"2024-05-18T07:03:12.117405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adjust_data_types(data_frame: pl.DataFrame) -> pl.DataFrame:\n    for column_name in data_frame.columns:\n        if column_name.endswith((\"P\", \"A\")):\n            data_frame = data_frame.with_columns(pl.col(column_name).cast(pl.Float64).alias(column_name))\n    return data_frame\n\ndef transform_string_columns(data: pd.DataFrame) -> pd.DataFrame:\n    for column in data.columns:\n        if data[column].dtype.name in ['object', 'string']:\n            data[column] = data[column].astype(\"string\").astype('category')\n            existing_categories = data[column].cat.categories\n            updated_categories = existing_categories.to_list() + [\"Unknown\"]\n            updated_type = pd.CategoricalDtype(categories=updated_categories, ordered=True)\n            data[column] = data[column].astype(updated_type)\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:12.120619Z","iopub.execute_input":"2024-05-18T07:03:12.121132Z","iopub.status.idle":"2024-05-18T07:03:12.130885Z","shell.execute_reply.started":"2024-05-18T07:03:12.1211Z","shell.execute_reply":"2024-05-18T07:03:12.129663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(adjust_data_types),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(adjust_data_types),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(adjust_data_types)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(adjust_data_types) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(adjust_data_types) ","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:12.132323Z","iopub.execute_input":"2024-05-18T07:03:12.132715Z","iopub.status.idle":"2024-05-18T07:03:30.363509Z","shell.execute_reply.started":"2024-05-18T07:03:12.132686Z","shell.execute_reply":"2024-05-18T07:03:30.362507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(adjust_data_types),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(adjust_data_types),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(adjust_data_types),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(adjust_data_types)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(adjust_data_types) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(adjust_data_types) ","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:30.366662Z","iopub.execute_input":"2024-05-18T07:03:30.367118Z","iopub.status.idle":"2024-05-18T07:03:30.422401Z","shell.execute_reply.started":"2024-05-18T07:03:30.367079Z","shell.execute_reply":"2024-05-18T07:03:30.421235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"person_1_features_group1 = train_person_1.groupby(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"max_main_occupation_income\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"any_selfemployed\")\n)\n\nperson_1_primary_applicant = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"applicant_housetype\"})\n\ncredit_bureau_aggregated = train_credit_bureau_b_2.groupby(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"max_overdue_payments\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"overdue_31_days\")\n)\n\nselected_static_features = [col for col in train_static.columns if col.endswith(('A', 'M'))]\nprint(selected_static_features)\n\nselected_cb_features = [col for col in train_static_cb.columns if col.endswith(('A', 'M'))]\nprint(selected_cb_features)\n\ncombined_data = train_basetable.join(\n    train_static.select([\"case_id\"] + selected_static_features), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"] + selected_cb_features), how=\"left\", on=\"case_id\"\n).join(\n    person_1_features_group1, how=\"left\", on=\"case_id\"\n).join(\n    person_1_primary_applicant, how=\"left\", on=\"case_id\"\n).join(\n    credit_bureau_aggregated, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:30.423688Z","iopub.execute_input":"2024-05-18T07:03:30.424038Z","iopub.status.idle":"2024-05-18T07:03:32.698945Z","shell.execute_reply.started":"2024-05-18T07:03:30.42401Z","shell.execute_reply":"2024-05-18T07:03:32.697773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_features_employment = test_person_1.groupby(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"max_main_occupation_income\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"any_selfemployed\")\n)\n\ntest_person_primary_applicant = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"applicant_housetype\"})\n\ntest_credit_bureau_overdue_details = test_credit_bureau_b_2.groupby(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"max_overdue_payments\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"overdue_31_days\")\n)\n\ntest_data_merged = test_basetable.join(\n    test_static.select([\"case_id\"] + selected_static_features), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"] + selected_cb_features), how=\"left\", on=\"case_id\"\n).join(\n    test_person_features_employment, how=\"left\", on=\"case_id\"\n).join(\n    test_person_primary_applicant, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_overdue_details, how=\"left\", on=\"case_id\"\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:32.700188Z","iopub.execute_input":"2024-05-18T07:03:32.700493Z","iopub.status.idle":"2024-05-18T07:03:32.715272Z","shell.execute_reply.started":"2024-05-18T07:03:32.700467Z","shell.execute_reply":"2024-05-18T07:03:32.714146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_case_ids = combined_data[\"case_id\"].unique().shuffle(seed=1)\ntraining_ids, test_ids = train_test_split(unique_case_ids, train_size=0.6, random_state=1)\nvalidation_ids, test_ids = train_test_split(test_ids, train_size=0.5, random_state=1)\n\npredicted_columns = [col for col in combined_data.columns if col[-1].isupper() and col[:-1].islower()]\nprint(predicted_columns)\n\ndef convert_to_pandas(case_ids: pl.Series) -> (pd.DataFrame, pd.DataFrame, pd.Series):\n    filtered_data = combined_data.filter(pl.col(\"case_id\").is_in(case_ids))\n    base_data = filtered_data[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas()\n    features_data = filtered_data[predicted_columns].to_pandas()\n    target_series = filtered_data[\"target\"].to_pandas()\n    return base_data, features_data, target_series\n\nbase_train, X_train, y_train = convert_to_pandas(training_ids)\nbase_valid, X_valid, y_valid = convert_to_pandas(validation_ids)\nbase_test, X_test, y_test = convert_to_pandas(test_ids)\n\nX_train = transform_string_columns(X_train)\nX_valid = transform_string_columns(X_valid)\nX_test = transform_string_columns(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:32.716532Z","iopub.execute_input":"2024-05-18T07:03:32.716854Z","iopub.status.idle":"2024-05-18T07:03:39.967951Z","shell.execute_reply.started":"2024-05-18T07:03:32.716814Z","shell.execute_reply":"2024-05-18T07:03:39.966543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:39.969557Z","iopub.execute_input":"2024-05-18T07:03:39.970722Z","iopub.status.idle":"2024-05-18T07:03:39.977726Z","shell.execute_reply.started":"2024-05-18T07:03:39.970677Z","shell.execute_reply":"2024-05-18T07:03:39.976354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.hist(X_train['annuity_780A'].dropna(), bins=50, color='blue', alpha=0.7)\nplt.title('Distribution of Annuity')\nplt.xlabel('Annuity Amount')\nplt.ylabel('Frequency')\n\nplt.subplot(1, 2, 2)\nplt.hist(X_train['credamount_770A'].dropna(), bins=50, color='green', alpha=0.7)\nplt.title('Distribution of Credit Amount')\nplt.xlabel('Credit Amount')\nplt.ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:39.979728Z","iopub.execute_input":"2024-05-18T07:03:39.980607Z","iopub.status.idle":"2024-05-18T07:03:40.848578Z","shell.execute_reply.started":"2024-05-18T07:03:39.980567Z","shell.execute_reply":"2024-05-18T07:03:40.847308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\nplt.plot(X_train['avgpmtlast12m_4525200A'].dropna(), label='Average Payment Last 12 Months')\nplt.plot(X_train['maxpmtlast3m_4525190A'].dropna(), label='Max Payment Last 3 Months')\nplt.plot(X_train['pmtaverage_3A'].dropna(), label='Average Payment')\nplt.title('Trend in Payment Amounts Over Time')\nplt.xlabel('Time')\nplt.ylabel('Amount')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:40.853754Z","iopub.execute_input":"2024-05-18T07:03:40.854141Z","iopub.status.idle":"2024-05-18T07:03:43.835599Z","shell.execute_reply.started":"2024-05-18T07:03:40.85411Z","shell.execute_reply":"2024-05-18T07:03:43.833919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.boxplot([X_train['avgoutstandbalancel6m_4187114A'].dropna(), X_train['maxoutstandbalancel12m_4187113A'].dropna()],\n            labels=['Avg Outstanding Balance Last 6m', 'Max Outstanding Balance Last 12m'])\nplt.title('Box Plot of Outstanding Balances')\nplt.ylabel('Amount')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:43.837425Z","iopub.execute_input":"2024-05-18T07:03:43.837867Z","iopub.status.idle":"2024-05-18T07:03:44.300326Z","shell.execute_reply.started":"2024-05-18T07:03:43.837807Z","shell.execute_reply":"2024-05-18T07:03:44.299128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.heatmap(X_train.isnull(), cbar=False)\nplt.title('Null Value Analysis')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:05:57.639322Z","iopub.execute_input":"2024-05-18T07:05:57.639767Z","iopub.status.idle":"2024-05-18T07:06:40.805529Z","shell.execute_reply.started":"2024-05-18T07:05:57.639734Z","shell.execute_reply":"2024-05-18T07:06:40.804276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.scatter(X_train['inittransactionamount_650A'], X_train['lastapprcredamount_781A'], alpha=0.5)\nplt.title('Initial vs. Last Approved Loan Amount Comparison')\nplt.xlabel('Initial Transaction Amount')\nplt.ylabel('Last Approved Credit Amount')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:44.309208Z","iopub.execute_input":"2024-05-18T07:03:44.309595Z","iopub.status.idle":"2024-05-18T07:03:45.034691Z","shell.execute_reply.started":"2024-05-18T07:03:44.309565Z","shell.execute_reply":"2024-05-18T07:03:45.033434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(14, 6))\nX_train['education_1103M'].value_counts().plot(kind='bar', ax=axes[0], color='skyblue')\naxes[0].set_title('Distribution of Education Levels')\naxes[0].set_ylabel('Frequency')\nX_train['maritalst_385M'].value_counts().plot(kind='bar', ax=axes[1], color='green')\naxes[1].set_title('Distribution of Marital Statuses')\naxes[1].set_ylabel('Frequency')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:45.036178Z","iopub.execute_input":"2024-05-18T07:03:45.036581Z","iopub.status.idle":"2024-05-18T07:03:45.727063Z","shell.execute_reply.started":"2024-05-18T07:03:45.036544Z","shell.execute_reply":"2024-05-18T07:03:45.725865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nX_train['lastrejectreason_759M'].value_counts().plot(kind='pie', autopct='%1.1f%%')\nplt.title('Distribution of Loan Rejection Reasons')\nplt.ylabel('') \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:45.728453Z","iopub.execute_input":"2024-05-18T07:03:45.728902Z","iopub.status.idle":"2024-05-18T07:03:46.103383Z","shell.execute_reply.started":"2024-05-18T07:03:45.728861Z","shell.execute_reply":"2024-05-18T07:03:46.10225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax1 = plt.subplots(figsize=(12, 6))\n\nax2 = ax1.twinx()\nax1.plot(X_train['annuitynextmonth_57A'], 'g-')\nax2.plot(X_train['totaldebt_9A'], 'b-')\n\nax1.set_xlabel('Time')\nax1.set_ylabel('Annuity Next Month', color='g')\nax2.set_ylabel('Total Debt', color='b')\n\nplt.title('Comparison of Annuity Payments and Total Debts Over Time')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:46.104687Z","iopub.execute_input":"2024-05-18T07:03:46.105028Z","iopub.status.idle":"2024-05-18T07:03:47.290208Z","shell.execute_reply.started":"2024-05-18T07:03:46.105Z","shell.execute_reply":"2024-05-18T07:03:47.288867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.scatter(X_train['totaldebt_9A'], X_train['credamount_770A'], alpha=0.5)\nplt.title('Scatter Plot of Debt vs. Loan Amount')\nplt.xlabel('Total Debt')\nplt.ylabel('Credit Amount')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:47.292215Z","iopub.execute_input":"2024-05-18T07:03:47.292651Z","iopub.status.idle":"2024-05-18T07:03:49.975548Z","shell.execute_reply.started":"2024-05-18T07:03:47.292613Z","shell.execute_reply":"2024-05-18T07:03:49.97427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\n\nX_train_c = X_train.copy()\nnon_numeric_columns = X_train_c.select_dtypes(exclude=[np.number]).columns\nfor column in non_numeric_columns:\n    X_train_c[column] = le.fit_transform(X_train_c[column].astype(str).fillna('missing'))\n\n\ncombined_data = X_train_c\ncombined_data['y_train'] = y_train\n\ncorr_matrix = combined_data.corr()\n\ntop_10_features = corr_matrix['y_train'].abs().sort_values(ascending=False).head(11).index\n\ntop_corr_matrix = combined_data[top_10_features].corr()\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(top_corr_matrix, annot=True, cmap='coolwarm', fmt=\".2f\", linewidths=.5)\nplt.title('Heatmap of Top Features Correlated with y_train Including Correlation Values')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:05:06.052934Z","iopub.execute_input":"2024-05-18T07:05:06.053361Z","iopub.status.idle":"2024-05-18T07:05:19.297459Z","shell.execute_reply.started":"2024-05-18T07:05:06.053327Z","shell.execute_reply":"2024-05-18T07:05:19.296318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_copy = X_train.copy()\ncategorical_columns = X_train.select_dtypes(include=['object']).columns\nnumeric_columns = X_train.select_dtypes(include=['number']).columns\n\nX_train_copy[categorical_columns] = X_train_copy[categorical_columns].fillna(\"missing\")\n\nX_train_copy[numeric_columns] = X_train_copy[numeric_columns].fillna(0)\nX_train_copy = X_train_copy[:]\nX_train_copy","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:49.976986Z","iopub.execute_input":"2024-05-18T07:03:49.977332Z","iopub.status.idle":"2024-05-18T07:03:50.635214Z","shell.execute_reply.started":"2024-05-18T07:03:49.977302Z","shell.execute_reply":"2024-05-18T07:03:50.633893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(X):\n    for col in X.columns:\n        if X[col].dtype == 'object' or X[col].dtype.name == 'category':\n            X[col] = X[col].astype('category').cat.codes\n        elif X[col].dtype == 'int64' or X[col].dtype == 'float64':\n            continue \n        else:\n            raise ValueError(f\"Column {col} cannot be processed due to incompatible type {X[col].dtype}\")\n    return X\n\nX_train_prepared = prepare_data(X_train_copy.copy())\nX_valid_prepared = prepare_data(X_train_copy.copy())\nX_test_prepared = prepare_data(X_train_copy.copy())\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_prepared)\nX_valid_scaled = scaler.transform(X_valid_prepared)\nX_test_scaled = scaler.transform(X_test_prepared)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:50.636503Z","iopub.execute_input":"2024-05-18T07:03:50.636871Z","iopub.status.idle":"2024-05-18T07:03:50.709908Z","shell.execute_reply.started":"2024-05-18T07:03:50.636841Z","shell.execute_reply":"2024-05-18T07:03:50.708847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ks = [3, 5, 7, 9, 11]\nmetrics = ['euclidean', 'manhattan', 'minkowski']\n\nresults = {}\n\nfor k in ks:\n    for metric in metrics:\n        knn = KNeighborsClassifier(n_neighbors=k, metric=metric)\n        knn.fit(X_train_scaled, y_train[:])\n\n        y_pred_prob = knn.predict_proba(X_valid_scaled)[:, 1]\n        auc_score = roc_auc_score(y_valid[:], y_pred_prob)\n        results[(k, metric)] = auc_score\n        print(f'AUC Score for k={k} with {metric} metric: {auc_score}')\n\nbest_k, best_metric = max(results, key=results.get)\nbest_auc = results[(best_k, best_metric)]\nprint(f'Best AUC Score {best_auc} achieved with k={best_k} and metric={best_metric}')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:03:50.711252Z","iopub.execute_input":"2024-05-18T07:03:50.711669Z","iopub.status.idle":"2024-05-18T07:04:12.318472Z","shell.execute_reply.started":"2024-05-18T07:03:50.711632Z","shell.execute_reply":"2024-05-18T07:04:12.317416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(X):\n    for col in X.columns:\n        if X[col].dtype == 'object' or X[col].dtype.name == 'category':\n            X[col] = X[col].astype('category').cat.codes\n        elif X[col].dtype.kind in 'biufc': \n            continue  \n        else:\n            raise ValueError(f\"Column {col} cannot be processed due to incompatible type {X[col].dtype}\")\n    return X\n\nbest_knn = KNeighborsClassifier(n_neighbors=best_k, metric=best_metric)\nbest_knn.fit(X_train_scaled, y_train[:])\n\ny_test_pred_prob = best_knn.predict_proba(X_test_scaled)[:, 1]\ntest_auc = roc_auc_score(y_test[:], y_test_pred_prob)\nprint(f'Test AUC: {test_auc}')\nX_submission = prepare_data(test_data_merged[predicted_columns].to_pandas()) \nX_submission_prepared = prepare_data(X_submission.copy())\nX_submission_scaled = scaler.transform(X_submission_prepared)\nX_submission_scaled_copy = np.nan_to_num(X_submission_scaled, nan=0.0)\ny_submission_pred = best_knn.predict_proba(X_submission_scaled_copy)[:, 1]\n\n\nsubmission = pd.DataFrame({\n    \"case_id\": test_data_merged[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\n\n# submission.to_csv(\"./submission.csv\")\nprint(submission)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:04:12.322469Z","iopub.execute_input":"2024-05-18T07:04:12.32288Z","iopub.status.idle":"2024-05-18T07:04:12.717305Z","shell.execute_reply.started":"2024-05-18T07:04:12.322846Z","shell.execute_reply":"2024-05-18T07:04:12.716342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(X):\n    cat_columns = X.select_dtypes(include=['category']).columns\n    for col in cat_columns:\n        X[col] = X[col].cat.codes \n    return X\n\nX_train_prepared = prepare_data(X_train.copy())\nX_valid_prepared = prepare_data(X_valid.copy())\nX_test_prepared = prepare_data(X_test.copy())\n\ndtrain = xgb.DMatrix(X_train_prepared, label=y_train, enable_categorical=True)\ndvalid = xgb.DMatrix(X_valid_prepared, label=y_valid, enable_categorical=True)\n\nparams = {\n    \"booster\": \"gbtree\",\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"max_depth\": 3,\n    \"eta\": 0.05,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.9,\n    \"n_estimators\": 1000,\n    \"verbosity\": 0,\n}\n\nevals_result = {}\ngbm = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=1000,\n    evals=[(dvalid, \"validation\")],\n    early_stopping_rounds=10,\n    evals_result=evals_result,\n    verbose_eval=50\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:04:12.718604Z","iopub.execute_input":"2024-05-18T07:04:12.720108Z","iopub.status.idle":"2024-05-18T07:04:37.48432Z","shell.execute_reply.started":"2024-05-18T07:04:12.720073Z","shell.execute_reply":"2024-05-18T07:04:37.483277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_for_prediction(X, base, gbm):\n    dmatrix = xgb.DMatrix(prepare_data(X.copy()))\n    y_pred = gbm.predict(dmatrix, iteration_range=(0, gbm.best_iteration + 1))  # `+1` because the upper limit is exclusive\n    base[\"score\"] = y_pred\n    return base\n\nbase_train = prepare_for_prediction(X_train, base_train, gbm)\nbase_valid = prepare_for_prediction(X_valid, base_valid, gbm)\nbase_test = prepare_for_prediction(X_test, base_test, gbm)\n\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}')\nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}')\nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:04:37.485815Z","iopub.execute_input":"2024-05-18T07:04:37.489367Z","iopub.status.idle":"2024-05-18T07:04:41.295235Z","shell.execute_reply.started":"2024-05-18T07:04:37.489331Z","shell.execute_reply":"2024-05-18T07:04:41.293872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}')\nprint(f'The stability score on the valid set is: {stability_score_valid}')\nprint(f'The stability score on the test set is: {stability_score_test}')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:04:41.296954Z","iopub.execute_input":"2024-05-18T07:04:41.297415Z","iopub.status.idle":"2024-05-18T07:04:42.381016Z","shell.execute_reply.started":"2024-05-18T07:04:41.297374Z","shell.execute_reply":"2024-05-18T07:04:42.37981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(X):\n    cat_columns = X.select_dtypes(include=['object']).columns\n    for col in cat_columns:\n        X[col] = X[col].astype('category').cat.codes\n    return X\n\nX_submission_prepared = prepare_data(X_submission.copy())\n\nX_submission_dmat = xgb.DMatrix(X_submission_prepared)\n\ny_submission_pred = gbm.predict(X_submission_dmat, iteration_range=(0, gbm.best_iteration + 1))\n\nsubmission = pd.DataFrame({\n    \"case_id\": test_data_merged[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\n\nsubmission.to_csv(\"./submission.csv\")\nprint(submission)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T07:04:42.382479Z","iopub.execute_input":"2024-05-18T07:04:42.382933Z","iopub.status.idle":"2024-05-18T07:04:42.4104Z","shell.execute_reply.started":"2024-05-18T07:04:42.382894Z","shell.execute_reply":"2024-05-18T07:04:42.409379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}