{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":178158327,"sourceType":"kernelVersion"},{"sourceId":178159521,"sourceType":"kernelVersion"},{"sourceId":33095,"sourceType":"modelInstanceVersion","modelInstanceId":27710},{"sourceId":33096,"sourceType":"modelInstanceVersion","modelInstanceId":27711}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!python /kaggle/usr/lib/script1/script1.py","metadata":{"_uuid":"e5f8f072-a3b5-4104-9e08-43a17ad6a013","_cell_guid":"8fad5fc6-c963-4313-b01c-34b86d44df49","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-05-19T17:45:17.081937Z","iopub.execute_input":"2024-05-19T17:45:17.082198Z","iopub.status.idle":"2024-05-19T17:47:57.287484Z","shell.execute_reply.started":"2024-05-19T17:45:17.082173Z","shell.execute_reply":"2024-05-19T17:47:57.286291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary\nData Loading and Preprocessing: Data is loaded and cleaned by dropping unnecessary columns and handling missing values.\nNormalization: Features are standardized to have a mean of 0 and a standard deviation of 1.\nModel Definition and Training: An ANN model with multiple layers, dropout, and batch normalization is defined and trained on the data.\nPrediction and Submission: Predictions are made on the test set and adjusted based on a specific condition before being saved for submission.","metadata":{}},{"cell_type":"code","source":"%%capture\n!python /kaggle/usr/lib/0_585/0_585.py","metadata":{"execution":{"iopub.status.busy":"2024-05-19T17:47:57.289657Z","iopub.execute_input":"2024-05-19T17:47:57.290365Z","iopub.status.idle":"2024-05-19T17:48:09.055017Z","shell.execute_reply.started":"2024-05-19T17:47:57.290328Z","shell.execute_reply":"2024-05-19T17:48:09.053871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport joblib\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import KNNImputer","metadata":{"execution":{"iopub.status.busy":"2024-05-19T17:48:09.056473Z","iopub.execute_input":"2024-05-19T17:48:09.056852Z","iopub.status.idle":"2024-05-19T17:48:11.97796Z","shell.execute_reply.started":"2024-05-19T17:48:09.056814Z","shell.execute_reply":"2024-05-19T17:48:11.977189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train,y,df_test=joblib.load('/kaggle/working/data.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-05-19T17:48:11.980359Z","iopub.execute_input":"2024-05-19T17:48:11.981093Z","iopub.status.idle":"2024-05-19T17:48:12.923969Z","shell.execute_reply.started":"2024-05-19T17:48:11.981057Z","shell.execute_reply":"2024-05-19T17:48:12.923161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_remove = [\n    'assignmentdate_238D', 'assignmentdate_4527235D', 'pmtaverage_4527227A',\n    'pmtcount_4527229L', 'assignmentdate_4955616D', 'contractssum_5085716L',\n    'dateofbirth_342D', 'max_num_group1_7', 'for3years_128L',\n    'for3years_504L', 'for3years_584L', 'formonth_118L', 'formonth_206L',\n    'formonth_535L', 'forquarter_1017L', 'forquarter_462L',\n    'forquarter_634L', 'fortoday_1092L', 'forweek_1077L', 'forweek_528L',\n    'forweek_601L', 'foryear_818L', 'pmtaverage_3A', 'pmtaverage_4955615A',\n    'pmtcount_4955617L', 'pmtcount_693L', 'responsedate_4917613D',\n    'riskassesment_940T', 'avglnamtstart24m_4525187A', 'clientscnt_136L',\n    'datelastinstal40dpd_247D', 'inittransactionamount_650A',\n    'interestrategrace_34L', 'lastdependentsnum_448L', 'lastotherinc_902A',\n    'lastotherlnsexpense_631A', 'lastrepayingdate_696D',\n    'maxannuity_4075009A', 'payvacationpostpone_4187118D',\n    'validfrom_1069D', 'max_credacc_actualbalance_314A',\n    'max_credacc_maxhisbal_375A', 'max_credacc_minhisbal_90A',\n    'max_credacc_transactions_402L', 'max_revolvingaccount_394A',\n    'max_amount_4917619A', 'max_deductiondate_4917603D', 'max_num_group1_4',\n    'max_annualeffectiverate_199L', 'max_annualeffectiverate_63L',\n    'max_interestrate_508L', 'max_prolongationcount_1120L',\n    'max_prolongationcount_599L', 'max_debtvalue_227A', 'max_credlmt_1052A',\n    'max_residualamount_127A', 'max_credlmt_228A',\n    'max_debtpastduevalue_732A', 'max_pmtdaysoverdue_1135P', 'max_dpd_550P',\n    'max_installmentamount_833A', 'max_totalamount_503A',\n    'max_contractdate_551D', 'max_pmts_date_1107D', 'max_num_group1_12',\n    'max_num_group2', 'max_installmentamount_644A', 'max_totalamount_881A',\n    'max_credquantity_984L', 'max_dpdmax_851P', 'max_overdueamountmax_950A',\n    'max_dpdmaxdatemonth_804T', 'max_dpdmaxdateyear_742T',\n    'max_instlamount_892A', 'max_numberofinstls_810L',\n    'max_residualamount_1093A', 'max_residualamount_3940956A',\n    'max_contractmaturitydate_151D', 'max_interesteffectiverate_369L',\n    'max_interestrateyearly_538L', 'max_pmtnumpending_403L',\n    'max_amtdebitincoming_4809443A', 'max_amtdepositbalance_4809441A',\n    'max_amtdepositoutgoing_4809442A', 'max_num_group1_8',\n    'max_birthdate_87D', 'max_childnum_185L', 'max_amount_416A',\n    'max_openingdate_313D', 'max_num_group1_10', 'max_contractenddate_991D',\n    'max_last180dayaveragebalance_704A', 'max_last180dayturnover_1134A',\n    'max_last30dayturnover_651A', 'max_openingdate_857D',\n    'max_num_group1_11'\n]\n\n# Drop the columns from df_train\ndf_train = df_train.drop(columns=columns_to_remove)\n\n# Drop the columns from df_test\ndf_test = df_test.drop(columns=columns_to_remove)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-19T17:48:12.92502Z","iopub.execute_input":"2024-05-19T17:48:12.925304Z","iopub.status.idle":"2024-05-19T17:48:14.40348Z","shell.execute_reply.started":"2024-05-19T17:48:12.925279Z","shell.execute_reply":"2024-05-19T17:48:14.402661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y)","metadata":{"execution":{"iopub.status.busy":"2024-05-19T17:48:14.404883Z","iopub.execute_input":"2024-05-19T17:48:14.405271Z","iopub.status.idle":"2024-05-19T17:48:14.690202Z","shell.execute_reply.started":"2024-05-19T17:48:14.405234Z","shell.execute_reply":"2024-05-19T17:48:14.689089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nimport joblib\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\n\n# Load data\n# df_train, y, df_test = joblib.load('/kaggle/working/data.pkl')\n\n# Drop unwanted columns\ndf_train = df_train.drop(columns=[\"WEEK_NUM\", 'target'], errors='ignore')\ndf_test = df_test.drop(columns=[\"WEEK_NUM\", 'target', 'case_id'], errors='ignore')\n\n# Replace missing values with mean\nimputer = SimpleImputer(strategy='mean')\ndf_train = pd.DataFrame(imputer.fit_transform(df_train), columns=df_train.columns)\ndf_test = pd.DataFrame(imputer.transform(df_test), columns=df_test.columns)\n\n# Normalize the data\nscaler = StandardScaler()\ndf_train = pd.DataFrame(scaler.fit_transform(df_train), columns=df_train.columns)\ndf_test = pd.DataFrame(scaler.transform(df_test), columns=df_test.columns)\n\n# Split training data for validation\nX_train, X_val, y_train, y_val = train_test_split(df_train, y, test_size=0.2, random_state=42)\n\n# ANN Model\ndef create_ann_model(input_shape):\n    model = Sequential([\n        Dense(32, activation='relu', input_shape=(input_shape,)),\n        BatchNormalization(),\n        Dropout(0.1),\n        Dense(20, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.1),\n        Dense(16, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.1),\n        Dense(8, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.1),\n        Dense(4, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.1),\n        Dense(1, activation='sigmoid')\n    ])\n    model.compile(optimizer=Adam(), loss='binary_crossentropy', metrics=['AUC'])\n    return model\n\n# Define input shape\ninput_shape_ann = df_train.shape[1]\n\n# Create and train the ANN model\nann_model = create_ann_model(input_shape_ann)\n\nprint(\"Training ANN model...\")\nann_model.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=10, batch_size=240)\n\n# Prediction and submission\ny_pred_ann = ann_model.predict(df_test).flatten()\n\n# Apply condition\ncondition = y_pred_ann < 0.977\n\n# Prepare submission\ndf_subm = pd.read_csv(\"/kaggle/working/sub.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm.loc[condition, 'score'] = (df_subm.loc[condition, 'score'] - 0.072).clip(0)\ndf_subm.to_csv(\"submission.csv\")\n\n# Cleanup\n# os.remove('data.pkl')\n# os.remove('/kaggle/working/data.pkl')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-19T18:18:05.219974Z","iopub.execute_input":"2024-05-19T18:18:05.220721Z","iopub.status.idle":"2024-05-19T18:21:45.236184Z","shell.execute_reply.started":"2024-05-19T18:18:05.220689Z","shell.execute_reply":"2024-05-19T18:21:45.235184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_subm)","metadata":{"execution":{"iopub.status.busy":"2024-05-19T18:23:43.899122Z","iopub.execute_input":"2024-05-19T18:23:43.899458Z","iopub.status.idle":"2024-05-19T18:23:43.910364Z","shell.execute_reply.started":"2024-05-19T18:23:43.899433Z","shell.execute_reply":"2024-05-19T18:23:43.909432Z"},"trusted":true},"execution_count":null,"outputs":[]}]}