{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-01T03:43:53.716313Z","iopub.execute_input":"2024-05-01T03:43:53.716543Z","iopub.status.idle":"2024-05-01T03:43:54.944881Z","shell.execute_reply.started":"2024-05-01T03:43:53.716518Z","shell.execute_reply":"2024-05-01T03:43:54.944250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import pandas as pd\nimport os\nimport numpy as np\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score, train_test_split, GridSearchCV\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import accuracy_score\n\n# Define functions for preprocessing and feature engineering\nclass PipelineTransformer:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df[col] = df[col].astype(int)\n            elif col in [\"date_decision\"]:\n                df[col] = pd.to_datetime(df[col])\n            elif col[-1] in (\"P\", \"A\"):\n                df[col] = df[col].astype(float)\n        return df\n\n    @staticmethod\n    def handle_dates(df):\n        date_cols = [col for col in df.columns if col[-1] == \"C\" and df[col].dtype == 'datetime64[ns]']\n        print(\"Date columns:\", date_cols)\n        for col in date_cols:\n            print(\"Data type of\", col, \":\", df[col].dtype)\n            df[col] = (df[col] - df[\"date_decision\"]).dt.days\n        df = df.drop(columns=[\"date_decision\", \"MONTH\"])\n        return df\n\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].isnull().mean()\n                if isnull > 0.8:\n                    df = df.drop(columns=[col])\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == object):\n                freq = df[col].nunique()\n                if (freq == 1) or (freq > 30):\n                    df = df.drop(columns=[col])\n\n        return df\n\nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        num_cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_mean = df[num_cols].mean().add_prefix(\"mean_\")\n        return expr_mean\n\n    @staticmethod\n    def date_expr(df):\n        date_cols = [col for col in df.columns if col.endswith(\"D\")]\n        expr_last = df[date_cols].tail(1).add_prefix(\"last_\")\n        return expr_last\n\n    @staticmethod\n    def str_expr(df):\n        str_cols = [col for col in df.columns if col[-1] == \"M\"]\n        expr_first = df[str_cols].head(1).add_prefix(\"first_\")\n        return expr_first\n\n    @staticmethod\n    def other_expr(df):\n        numeric_cols = df.select_dtypes(include=['float'])\n        string_cols = df.select_dtypes(include=['object'])\n\n        exprs = []\n\n        if not numeric_cols.empty:\n            exprs.append(numeric_cols.mean().add_prefix(\"mean_numeric_\"))\n\n        if not string_cols.empty:\n            exprs.append(string_cols.iloc[:, 0].add_prefix(\"first_string_\"))\n\n        return exprs\n\n    @staticmethod\n    def count_expr(df):\n        count_cols = [col for col in df.columns if \"num_group\" in col]\n        expr_first = df[count_cols].head(1).add_prefix(\"first_\")\n        return expr_first\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = (\n            Aggregator.num_expr(df)\n            + Aggregator.date_expr(df)\n            + Aggregator.str_expr(df)\n            + Aggregator.other_expr(df)\n            + [Aggregator.count_expr(df)]\n        )\n\n        return pd.concat(exprs, axis=1)\n\n# Define function to read and preprocess CSV files\ndef preprocess_csv(file_path, is_train=True):\n    df = pd.read_csv(file_path)\n    print(f\"CSV file '{file_path}' read successfully.\")\n    \n    df[\"date_decision\"] = pd.to_datetime(df[\"date_decision\"])\n    df = PipelineTransformer.set_table_dtypes(df)\n    df = PipelineTransformer.handle_dates(df)\n    df = PipelineTransformer.filter_cols(df)\n\n    if is_train:\n        if 'target' in df.columns:\n            df['target'] = df['target'].astype(int)\n            print(\"Distribution of target variable ('target'):\")\n            print(df['target'].value_counts())\n        else:\n            print(\"Warning: 'target' column not found in the dataframe.\")\n    else:\n        df['target'] = 0  \n\n    return df\n\n# Define function to merge CSV files\ndef merge_files(input_files, output_file):\n    all_data = pd.concat((pd.read_csv(file) for file in input_files), ignore_index=True)\n    all_data = all_data.groupby('case_id').sum().reset_index()\n    all_data.to_csv(output_file, index=False)\n    print(f\"Merged data saved to '{output_file}'\")\n    return all_data\n\n# Define function to save modified CSV\ndef save_modified_csv(df, file_path):\n    df.to_csv(file_path, index=False)\n    print(f\"Modified data saved to '{file_path}'\")\n    \n# Define function to apply SMOTE to balance the data\ndef apply_smote(df, target_column, balance_ratio, random_state=3270):\n    X = df.drop(columns=[target_column])\n    y = df[target_column]\n\n    imbalance_ratio = y.value_counts()[0] / y.value_counts()[1]\n        \n    smote = SMOTE(sampling_strategy=balance_ratio, random_state=random_state)\n    X_resampled, y_resampled = smote.fit_resample(X, y)\n\n    df_resampled = pd.concat([X_resampled, y_resampled], axis=1)\n    return df_resampled, imbalance_ratio\n\ndef perform_grid_search(X, y):\n    param_grid = {\n        'solver': ['adam', 'sgd'],\n        'hidden_layer_sizes': [(50,), (100,), (50, 50), (100, 50)],\n        'activation': ['logistic', 'relu'],\n        'max_iter': [1000, 2000],\n    }\n\n    mlp = MLPClassifier()\n\n    grid_search = GridSearchCV(estimator=mlp, param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)\n    grid_search.fit(X, y)\n\n    results = pd.DataFrame(grid_search.cv_results_)\n    top_10_results = results[['param_solver', 'param_hidden_layer_sizes', 'param_activation', 'param_max_iter', 'mean_test_score']].sort_values(by='mean_test_score', ascending=False).head(10)\n\n    return top_10_results\n\ndef test(X_scaled, y):\n    HIDDEN_LAYERS = (50,)\n    SOLVER = 'adam'\n    MAX_ITER = 1000\n    ACTIVATION = 'logistic'\n    SEED = 3270\n\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n    total_correct = 0\n    total_wrong = 0\n    total_correct_0 = 0\n    total_wrong_0 = 0\n    total_correct_1 = 0\n    total_wrong_1 = 0\n\n    for i, (train_index, test_index) in enumerate(skf.split(X_scaled, y)):\n        X_train, X_test = X_scaled[train_index], X_scaled[test_index]\n        y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n\n        mlp_model = MLPClassifier(solver=SOLVER, hidden_layer_sizes=HIDDEN_LAYERS,\n                                   max_iter=MAX_ITER, activation=ACTIVATION, random_state=SEED)\n        mlp_model.fit(X_train, y_train)\n\n        y_pred = mlp_model.predict(X_test)\n        correct_predictions = sum(y_pred == y_test)\n        wrong_predictions = len(y_test) - correct_predictions\n        accuracy = correct_predictions / len(y_test)\n\n        correct_predictions_0 = sum((y_pred == y_test) & (y_test == 0))\n        wrong_predictions_0 = sum((y_pred != y_test) & (y_test == 0))\n        correct_predictions_1 = sum((y_pred == y_test) & (y_test == 1))\n        wrong_predictions_1 = sum((y_pred != y_test) & (y_test == 1))\n\n        total_correct += correct_predictions\n        total_wrong += wrong_predictions\n        total_correct_0 += correct_predictions_0\n        total_wrong_0 += wrong_predictions_0\n        total_correct_1 += correct_predictions_1\n        total_wrong_1 += wrong_predictions_1\n\n        print(f'Processing split {i+1}...')\n        print(f'Creating neural network model...')\n        print(f'Neural network model created successfully.')\n        print(f'Split {i+1}: Correct predictions: {correct_predictions}, '\n              f'Wrong predictions: {wrong_predictions}, Accuracy: {accuracy:.4f}')\n        print(f'Predicting target \"0\": Correct predictions: {correct_predictions_0}, '\n              f'Wrong predictions: {wrong_predictions_0}')\n        print(f'Predicting target \"1\": Correct predictions: {correct_predictions_1}, '\n              f'Wrong predictions: {wrong_predictions_1}')\n\n    total_splits = skf.get_n_splits(X_scaled, y)\n    print(f'\\nGrand Total:')\n    print(f'Total Correct Predictions: {total_correct}, Total Wrong Predictions: {total_wrong}')\n    print(f'Total Correct Predictions for target \"0\": {total_correct_0}, '\n          f'Total Wrong Predictions for target \"0\": {total_wrong_0}')\n    print(f'Total Correct Predictions for target \"1\": {total_correct_1}, '\n          f'Total Wrong Predictions for target \"1\": {total_wrong_1}')\n\n    mlp_scores = cross_val_score(MLPClassifier(solver=SOLVER, hidden_layer_sizes=HIDDEN_LAYERS,\n                                               max_iter=MAX_ITER, activation=ACTIVATION,\n                                               random_state=SEED), X_scaled, y, cv=skf, scoring=\"accuracy\")\n    print(f'\\nMLP Scores: {mlp_scores.mean():.2f} ± {mlp_scores.std():.2f}')\n\n    \ndef remove_columns_by_index(csv_file, indices):\n    # Read the CSV file\n    df = pd.read_csv(csv_file)\n    \n    # Remove columns by indices\n    columns_to_remove = df.columns[indices]\n    df.drop(columns=columns_to_remove, inplace=True)\n    \n    # Write the modified DataFrame back to the CSV file\n    df.to_csv(csv_file, index=False)\n\n\ndef main():\n    print(\"Starting main function...\")\n\n    train_input_files = [\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_a_1_0.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_a_1_1.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_a_1_2.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_a_1_3.csv'\n    ]\n    test_input_files = [\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_credit_bureau_a_1_0.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_credit_bureau_a_1_1.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_credit_bureau_a_1_2.csv',\n        '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_credit_bureau_a_1_3.csv'\n    ]\n\n    train_merged_file = 'train.csv'\n    test_merged_file = 'test.csv'\n\n    print(\"Merging training data...\")\n    train_data = merge_files(train_input_files, train_merged_file)\n\n    print(\"Merging test data...\")\n    test_data = merge_files(test_input_files, test_merged_file)\n\n    print(\"Preprocessing training data...\")\n    df_train = preprocess_csv(train_merged_file, is_train=True)\n\n    print(\"Preprocessing test data...\")\n    df_test = preprocess_csv(test_merged_file, is_train=False)\n\n    save_modified_csv(df_train, 'train_modified.csv')\n    save_modified_csv(df_test, 'test_modified.csv')\n    print(\"Modified data saved.\")\n\n    df_train_modified = pd.read_csv('train_modified.csv')\n\n    df_train_sampled = df_train_modified.sample(frac=0.02, random_state=2370)\n\n    print(\"Distribution of target variable ('target') in the sampled train data:\")\n    print(df_train_sampled['target'].value_counts())\n\n    print(\"Encoding categorical features...\")\n    df_train_encoded = pd.get_dummies(df_train_sampled)\n\n    print(\"Applying SMOTE to balance the data...\")\n    balance_ratio = 1\n    df_train_balanced, imbalance_ratio_before = apply_smote(df_train_encoded, 'target', balance_ratio)\n\n    save_modified_csv(df_train_balanced, 'train_balanced.csv')\n\n    print(\"Distribution of target variable ('target') in the balanced train data:\")\n    print(df_train_balanced['target'].value_counts())\n\n    imbalance_ratio_after = df_train_balanced['target'].value_counts()[0] / df_train_balanced['target'].value_counts()[1]\n    print(f\"Imbalance ratio after SMOTE: {imbalance_ratio_after}\")\n\n    print(\"Preparing data for hyperparameter tuning...\")\n    X_train = df_train_balanced.drop(columns=['target'])\n    y_train = df_train_balanced['target']\n\n    numeric_features = X_train.select_dtypes(include=['float64']).columns\n    categorical_features = X_train.select_dtypes(include=['object']).columns\n\n    column_transformer = ColumnTransformer([\n        ('numeric', StandardScaler(), numeric_features),\n        ('categorical', OneHotEncoder(), categorical_features)\n    ])\n\n    pipeline = Pipeline([\n        ('preprocessor', column_transformer),\n    ])\n\n    X_scaled = pipeline.fit_transform(X_train)\n\n    #top_10_params = perform_grid_search(X_scaled, y_train)\n    #print(\"Top 10 parameter sets and their accuracies:\")\n    #print(top_10_params)\n    \n    # Assuming indices to remove are stored in an array called 'indices_to_remove'\n    indices_to_remove = [28,30,26,4,31,6,24,29,20,22]  # Example: Remove columns with the specified indices\n\n    # Remove columns from CSV files before testing\n    remove_columns_by_index('/kaggle/working/train_balanced.csv', indices_to_remove)\n    remove_columns_by_index('/kaggle/working/test_modified.csv', indices_to_remove)\n\n    print(\"Testing the model...\")\n    \n    test(X_scaled, y_train)\n\n    print(\"Main function completed successfully.\")\n\nif __name__ == \"__main__\":\n    main()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T03:26:53.422771Z","iopub.execute_input":"2024-05-01T03:26:53.423034Z","iopub.status.idle":"2024-05-01T03:26:54.953693Z","shell.execute_reply.started":"2024-05-01T03:26:53.423007Z","shell.execute_reply":"2024-05-01T03:26:54.952888Z"}}}]}