{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting using only some features\n\nThe purpose of this notebook is to create a model based on the features which has the highest correlation with target and then remove the ones that has correlation above 0.75 between them","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LogisticRegression, LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get only the first 1M to get the correlation\ntrain_data_first = pd.read_csv(\n    input_path / 'train_data.csv',\n    index_col='customer_ID',\n    nrows=1_000_000)\n\ntrain_labels_first = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID', nrows=1_000_000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get only the labels of the customers of the first 1M rows of the train data\ntrain_labels_first = train_labels_first[train_labels_first.index.isin(train_data_first.index)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We are going to use only the last month\nlast_month_train_data_first = train_data_first.groupby('customer_ID').tail(1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_train_data_first = last_month_train_data_first.merge(train_labels_first, on='customer_ID', \n                                               how='inner', validate='one_to_one')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we get the first 20\nbest_20_pred = last_month_train_data_first.corr()['target'].abs().sort_values(ascending=False).index[1:21]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cor_matrix = last_month_train_data_first[best_20_pred].corr()\nupper_tri = cor_matrix.where(np.triu(np.ones(cor_matrix.shape),k=1).astype(bool))\nto_drop = [column for column in upper_tri.columns if any(upper_tri[column] > 0.75)]\n\npredictors = list(set(best_20_pred).difference(set(to_drop)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\n    input_path / 'train_data.csv',\n    usecols=predictors+['customer_ID'])\n\ntrain_labels = pd.read_csv(input_path / 'train_labels.csv', index_col='customer_ID')\n\nlast_month_train_data = train_data.groupby('customer_ID').tail(1)\n\nlast_month_train_data = last_month_train_data.merge(train_labels, on='customer_ID', how='inner',\n                                                    validate='one_to_one')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestRegressor()\nrf.fit(last_month_train_data[predictors].fillna(-999), \n       last_month_train_data['target'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\n    input_path / 'test_data.csv',\n    usecols=predictors+['customer_ID'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data = test_data.groupby('customer_ID').tail(1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data['prediction'] = rf.predict(last_month_test_data[predictors].fillna(-999))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_month_test_data[['customer_ID', 'prediction']].to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}