{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609},{"sourceId":98788536,"sourceType":"kernelVersion"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"edf82166-93fd-4d4e-8c99-75e29832b14d","_cell_guid":"0481960a-8cef-4cf7-a0d6-3948812d2dfb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-13T08:27:04.186108Z","iopub.execute_input":"2024-04-13T08:27:04.186812Z","iopub.status.idle":"2024-04-13T08:27:04.258681Z","shell.execute_reply.started":"2024-04-13T08:27:04.186776Z","shell.execute_reply":"2024-04-13T08:27:04.257074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\n","metadata":{"_uuid":"f21225ac-2193-4eae-bc3a-b5c527ba3d4e","_cell_guid":"f3c28cb3-a174-48c9-9ddd-421d50ef5367","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-13T08:27:04.260789Z","iopub.execute_input":"2024-04-13T08:27:04.261279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = pd.read_feather('/kaggle/input/amexfeather/train_data.ftr')\ntest = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n","metadata":{"_uuid":"c3172903-40fa-482a-8713-2352da76e7c3","_cell_guid":"1161447e-59a7-4349-9923-a11160e87498","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntrain.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntrain.isna().any().any(), train.customer_ID.duplicated().any()","metadata":{"_uuid":"a52297c6-1700-4b95-9fad-733ed9b02e5c","_cell_guid":"4655b924-c69d-4bc1-8ec5-f897a06410cc","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfrom sklearn.preprocessing import MinMaxScaler\n\n# Assuming df is your DataFrame and 'date_column' is your date column in datetime format\ntrain['S_2'] = pd.to_datetime(train['S_2'])\ntest['S_2'] = pd.to_datetime(test['S_2'])\n# Convert the datetime to Unix timestamp (seconds since Epoch) and then to integer\ntrain['S_2'] = train['S_2'].astype('int64') // 10**9\ntest['S_2'] = test['S_2'].astype('int64') // 10**9\n\n# Assuming 'date_as_int' is the column you want to normalize\nscaler = MinMaxScaler()\n\n# Reshape your data to fit the scaler\ndate_int_scaled = scaler.fit_transform(train[['S_2']])\ndate_int_scaledt = scaler.fit_transform(test[['S_2']])\n\n# Replace the original column with the scaled values\ntrain['S_2'] = date_int_scaled\ntest['S_2'] = date_int_scaledt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nmissing_values = train.isna().sum()\n\n# Sort columns based on missing values count\nsorted_columns = missing_values.sort_values(ascending=False)\n\n# Get names of top 20 columns with most NaN values\ntop_30_columns_with_most_nans = sorted_columns.head(10).index.tolist()\n# Remove columns with most NaN values\ntrain = train.drop(columns=top_30_columns_with_most_nans)\ntest = test.drop(columns=top_30_columns_with_most_nans)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\nmiss = train.isna().sum().sort_values(ascending=False).reset_index(drop=False).rename({0:'missing_count'}, axis=1)\nmiss['miss_ratio'] = (miss['missing_count'] / train.shape[0]) * 100\n\nsns.barplot(data=miss[:25], x= 'miss_ratio', y='index', palette='YlOrBr_r', linewidth=0.7, edgecolor=\".2\")\nplt.title('Missing Value Ratios Per Column')\nplt.ylabel('Column')\nplt.xlabel('Percent of Missing Values')\ndel miss\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfrom sklearn.model_selection import train_test_split\n\n# Split the data into features and target\nX = train.drop(['target', 'customer_ID'], axis=1)\ny = train['target']\nXt=test.drop('customer_ID', axis=1)\n# Split the data into training and test sets\nXtrain, Xtest, Ytrain, Ytest = train_test_split(X, y, test_size=0.20, random_state=42)\n\n","metadata":{"_uuid":"f98a8cb3-5147-488b-9c53-55198d68deb4","_cell_guid":"a94ddcd8-5f85-4196-8b37-14b2cc419243","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfrom sklearn.impute import KNNImputer\n\n# Assuming 'train' is your DataFrame with missing values\nimputer = KNNImputer(n_neighbors=5)  # You can adjust the number of neighbors as needed\ntrain_imputed = pd.DataFrame(imputer.fit_transform(train), columns=train.columns)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nimport numpy as np\nfrom keras.layers import Input, Dense\nfrom keras.models import Model\n\n# Define the size of our encoded representations\nencoding_dim = 32  # This is a hyperparameter you'll need to tune for your specific case\ninput_shape = Xtrain.shape[1]  # This gives you the number\n\n# Input placeholder\ninput_data = Input(shape=(input_shape,))  # 'input_shape' is the number of features in your data\n\n# \"Encoded\" representation of the input\nencoded = Dense(encoding_dim, activation='relu')(input_data)\n\n# \"Decoded\" reconstruction of the input\ndecoded = Dense(input_shape, activation='sigmoid')(encoded)\n\n# This model maps an input to its reconstruction\nautoencoder = Model(input_data, decoded)\n\n# This model maps an input to its encoded representation\nencoder = Model(input_data, encoded)\n\n# Compile the autoencoder\nautoencoder.compile(optimizer='adam', loss='binary_crossentropy')\n\n# Prepare your data, replacing NaNs with zeros (or use another imputation strategy here)\nXtrain = Xtrain.fillna(0)\n\n# Train the autoencoder\nautoencoder.fit(Xtrain, Xtrain,\n                epochs=50,\n                batch_size=256,\n                shuffle=True,\n                validation_data=(Xtest.fillna(0), Xtest.fillna(0)))\n\n# Now you can use the encoder to transform data with missing values to the encoded space\nencoded_data = encoder.predict(Xtest.fillna(0))\n\n# If you want, you can use the autoencoder to predict (impute) the missing values\nimputed_data = autoencoder.predict(Xtest.fillna(0))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nimport lightgbm as lgb\ntrain_data = lgb.Dataset(Xtrain, label=Ytrain)\ntest_data = lgb.Dataset(Xtest, label=Ytest, reference=train_data)\n# Assuming Xtrain, Xtest, Ytrain, Ytest are your training and testing datasets\nparams = {\n    'objective': 'binary',  # Adjust according to your task\n    'metric': 'binary_logloss',  # Can be adjusted based on your objective\n    'boosting': 'gbdt',\n    'num_leaves': 15,  # Reduced from 31 to make the model lighter\n    'learning_rate': 0.05,\n    'bagging_fraction': 0.8,  # Use 80% of data for training to speed up training and reduce over-fitting\n    'bagging_freq': 5,  # Perform bagging every 5 iterations\n    'feature_fraction': 0.8,  # Use 80% of features to speed up training and reduce over-fitting\n    'min_data_in_leaf': 100,  # Can help with memory usage\n    'verbose': -1  # Less verbose\n}\n\n# Train the model with the updated parameters\ngbm = lgb.train(params,\n                train_data,\n                num_boost_round=20,\n                valid_sets=[test_data])\n","metadata":{"_uuid":"e0b8bbc0-c05d-4a16-bb93-a97b40811f40","_cell_guid":"fd079968-0da0-4912-a6f9-e2f0097438f0","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n#This is the metric we are asked to use\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.to_frame(name='prediction')\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"_uuid":"954630ec-3ab1-4aa0-8d0f-7401127d5287","_cell_guid":"1117b32f-ca33-48db-a36e-49af658c441a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ny_pred_prob = gbm.predict(Xtest, num_iteration=gbm.best_iteration)\ny_pred_df = pd.DataFrame(y_pred_prob, columns=['prediction'])\ny_true_df = Ytest.reset_index(drop=True) # Make sure Ytest is in the correct DataFrame format\n\n# Calculate the custom metric\ncustom_metric_score = amex_metric(y_true_df, y_pred_df)\nprint(\"Custom Metric Score:\", custom_metric_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ny_predict_prob = gbm.predict(Xt, num_iteration=gbm.best_iteration)\n\n","metadata":{"_uuid":"e509720f-6daf-469f-8ffa-9848f556584b","_cell_guid":"d463443d-cc33-425d-81da-210b88358dd1","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nsubmission_df = pd.DataFrame({\n    'ID': test['customer_ID'],  # Replace 'ID' with the actual ID column name for the competition\n    'prediction': y_predict_prob  # Replace 'prediction' with the actual prediction column name\n})\nsubmission_df.to_csv('submission.csv', index=False)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}