{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport time\nimport warnings\nimport gc\ngc.collect()\nimport os\nfrom six.moves import urllib\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\nwarnings.filterwarnings('ignore')\n%matplotlib inline\nplt.style.use('seaborn')\nfrom scipy.stats import norm, skew\nfrom sklearn.preprocessing import StandardScaler\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ee84b297643612e9330a4479889ac31dbd8ccc3"},"cell_type":"code","source":"#Add All the Models Libraries\n\n# Scalers\nfrom sklearn.utils import shuffle\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.pipeline import FeatureUnion\n\n# Models\n\nfrom sklearn.linear_model import Lasso\nfrom sklearn.metrics import mean_squared_log_error,mean_squared_error, r2_score,mean_absolute_error\n\nfrom sklearn.model_selection import train_test_split #training and testing data split\nfrom sklearn import metrics #accuracy measure\nfrom sklearn.metrics import confusion_matrix #for confusion matrix\nfrom scipy.stats import reciprocal, uniform\n\nfrom sklearn.model_selection import StratifiedKFold, RepeatedKFold\n\n# Cross-validation\nfrom sklearn.model_selection import KFold #for K-fold cross validation\nfrom sklearn.model_selection import cross_val_score #score evaluation\nfrom sklearn.model_selection import cross_val_predict #prediction\nfrom sklearn.model_selection import cross_validate\n\n# GridSearchCV\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RandomizedSearchCV\n\n#Common data processors\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn import feature_selection\nfrom sklearn import model_selection\nfrom sklearn import metrics\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.utils import check_array\nfrom scipy import sparse","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"309b8e771cbb5d3f5fe0b8c74dfef632e63e04a0"},"cell_type":"code","source":"#memory usage reduction\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40b87851bf7abd04564336df3efec653b9ded400"},"cell_type":"code","source":"train=reduce_mem_usage(pd.read_csv('../input/train.csv',parse_dates=[\"first_active_month\"]))\ntest=reduce_mem_usage(pd.read_csv('../input/test.csv',parse_dates=[\"first_active_month\"]))\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10ea9324ece77807d103313949206c919b95a98c"},"cell_type":"code","source":"import plotly\nimport plotly.graph_objs as go\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\ninit_notebook_mode(connected=True)    #THIS LINE IS MOST IMPORTANT AS THIS WILL DISPLAY PLOT ON \n#NOTEBOOK WHILE KERNEL IS RUNNING\n\nactive_month_series_traindf = train['first_active_month'].value_counts()\nactive_month_series_testdf = test['first_active_month'].value_counts()\n\ntrace0 = go.Scatter(\n        x = active_month_series_traindf.index,\n        y = active_month_series_traindf.values,\n        name = 'train set')\n\ntrace1 = go.Scatter(\n        x = active_month_series_testdf.index,\n        y = active_month_series_testdf.values,\n        name = 'test set')\n\nplotly.offline.iplot({\n    \"data\": [trace0, trace1],\n    \"layout\": go.Layout(title=\"First Active month in Train & Test data\")\n})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0eaca7a76a2a6e7a2353b9ea4936a8e14a253ceb"},"cell_type":"code","source":"def aggregate_transaction_hist(trans, prefix):  \n        \n    agg_func = {\n        'purchase_date' : ['max','min'],\n        'month_diff' : ['mean', 'min', 'max', 'var'],\n        'weekend' : ['sum', 'mean'],\n        'authorized_flag': ['sum', 'mean'],\n        'category_1': ['sum','mean', 'max','min'],\n        'purchase_amount': ['sum', 'mean', 'max', 'min', 'std'],\n        'installments': ['sum', 'mean', 'max', 'min', 'std'],  \n        'month_lag': ['max','min','mean','var'],\n        'card_id' : ['size'],\n        'month': ['nunique'],\n        'hour': ['nunique'],\n        'weekofyear': ['nunique'],\n        'dayofweek': ['nunique'],\n        'year': ['nunique'],\n        'subsector_id': ['nunique'],\n        'merchant_category_id' : ['nunique']\n    }\n    \n    agg_trans = trans.groupby(['card_id']).agg(agg_func)\n    agg_trans.columns = [prefix + '_'.join(col).strip() \n                           for col in agg_trans.columns.values]\n    agg_trans.reset_index(inplace=True)\n    \n    df = (trans.groupby('card_id')\n          .size()\n          .reset_index(name='{}transactions_count'.format(prefix)))\n    \n    agg_trans = pd.merge(df, agg_trans, on='card_id', how='left')\n    \n    return agg_trans","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ec05654353a6bbbfc30e5d1537f481752e5c59d"},"cell_type":"markdown","source":"Noe i will try to extract feature from Tansactions"},{"metadata":{"trusted":true,"_uuid":"413e4d7e6742bbf7c8c1ca77991996f0641f1bdb"},"cell_type":"code","source":"transactions=reduce_mem_usage(pd.read_csv('../input/historical_transactions.csv'))\ntransactions.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e926cf3c3030ea5cd781192724c5dc3d5435bf0"},"cell_type":"code","source":"transactions['authorized_flag'] = transactions['authorized_flag'].map({'Y': 1, 'N': 0})\ntransactions['category_1'] = transactions['category_1'].map({'Y': 1, 'N': 0})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2946ebda461e22a79c680c48a07d81921fe983f2"},"cell_type":"code","source":"#feature engineering adding new features\n# Now extract the month, year, day, weekday\ntrain[\"month\"] = train[\"first_active_month\"].dt.month\ntrain[\"year\"] = train[\"first_active_month\"].dt.year\ntrain['week'] = train[\"first_active_month\"].dt.weekofyear\ntrain['dayofweek'] = train['first_active_month'].dt.dayofweek\ntrain['days'] = (datetime.date(2018, 2, 1) - train['first_active_month'].dt.date).dt.days\ntrain['quarter'] = train['first_active_month'].dt.quarter\ntrain['is_month_start'] = train['first_active_month'].dt.is_month_start\n\n#Interaction Variables\ntrain['days_feature1'] = train['days'] * train['feature_1']\ntrain['days_feature2'] = train['days'] * train['feature_2']\ntrain['days_feature3'] = train['days'] * train['feature_3']\n\ntest[\"month\"] = test[\"first_active_month\"].dt.month\ntest[\"year\"] = test[\"first_active_month\"].dt.year\ntest['week'] = test[\"first_active_month\"].dt.weekofyear\ntest['dayofweek'] = test['first_active_month'].dt.dayofweek\ntest['days'] = (datetime.date(2019, 1, 30) - test['first_active_month'].dt.date).dt.days\ntest['quarter'] = test['first_active_month'].dt.quarter\ntest['is_month_start'] = test['first_active_month'].dt.is_month_start\n\n#Interaction Variables\ntest['days_feature1'] = test['days'] * train['feature_1']\ntest['days_feature2'] = test['days'] * train['feature_2']\ntest['days_feature3'] = test['days'] * train['feature_3']\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae932abe8d28174756504c26f57c1c20474ecdf8"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f643626bafbcade031e077459b50652087f4bd6a"},"cell_type":"code","source":"transactions.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"010fc2baf679e54b811fac37675d55c08ff396b3"},"cell_type":"code","source":"transactions['category_3'].value_counts()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dea9d3db3fb69f58eac29a3e103de57a506e4f13"},"cell_type":"code","source":"#Feature Engineering - Adding new features inspired by Chau's first kernel\ntransactions['purchase_date'] = pd.to_datetime(transactions['purchase_date'])\ntransactions['year'] = transactions['purchase_date'].dt.year\ntransactions['weekofyear'] = transactions['purchase_date'].dt.weekofyear\ntransactions['month'] = transactions['purchase_date'].dt.month\ntransactions['dayofweek'] = transactions['purchase_date'].dt.dayofweek\ntransactions['weekend'] = (transactions.purchase_date.dt.weekday >=5).astype(int)\ntransactions['hour'] = transactions['purchase_date'].dt.hour \ntransactions['quarter'] = transactions['purchase_date'].dt.quarter\ntransactions['is_month_start'] = transactions['purchase_date'].dt.is_month_start\ntransactions['month_diff'] = ((datetime.datetime.today() - transactions['purchase_date']).dt.days)//30\ntransactions['month_diff'] += transactions['month_lag']\n\n#impute missing values - This is now excluded.\ntransactions['category_2'] = transactions['category_2'].fillna(1.0,inplace=True)\ntransactions['category_3'] = transactions['category_3'].fillna('A',inplace=True)\ntransactions['merchant_id'] = transactions['merchant_id'].fillna('M_ID_00a6ca8a8a',inplace=True)\n\ntransactions['category_3'] = transactions['category_3'].map({'A':0, 'B':1, 'C':2})\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecb22717595ea3fa36b7cb20d0bbac97793e8e73"},"cell_type":"code","source":"\nagg_func = {\n        'mean': ['mean'],\n    }\nfor col in ['category_2','category_3']:\n    transactions[col+'_mean'] = transactions['purchase_amount'].groupby(transactions[col]).agg('mean')\n    transactions[col+'_max'] = transactions['purchase_amount'].groupby(transactions[col]).agg('max')\n    transactions[col+'_min'] = transactions['purchase_amount'].groupby(transactions[col]).agg('min')\n    transactions[col+'_var'] = transactions['purchase_amount'].groupby(transactions[col]).agg('var')\n    agg_func[col+'_mean'] = ['mean']\n    \ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b5804d04390c7fea089e5af3ad8339eae68b6a5"},"cell_type":"code","source":"merge_trans = aggregate_transaction_hist(transactions, prefix='hist_')\ndel transactions\ngc.collect()\ntrain = pd.merge(train, merge_trans, on='card_id',how='left')\ntest = pd.merge(test, merge_trans, on='card_id',how='left')\ndel merge_trans\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7025eff6fa9260e8311e77f130ab36577dd74575"},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7258e1a3a028a886d390e3e5b65dd737f2eb6566"},"cell_type":"code","source":"y=train[\"target\"]\ntrain.drop(\"target\",axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"096ea710b7570f6f81a8971e34e73b6b9c9aed2f"},"cell_type":"code","source":"train.drop([\"first_active_month\",\"card_id\",\"hist_purchase_date_max\", \"hist_purchase_date_min\"],axis=1,inplace=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3aebd2cd06f751dc5446e50559a742274d73c874"},"cell_type":"code","source":"import lightgbm as lgb\n\nd_train=lgb.Dataset(train,label=y)\nparams = {'num_leaves': 31,\n         'min_data_in_leaf': 27, \n         'objective':'regression',\n         'max_depth': -1,\n         'learning_rate': 0.015,\n         \"boosting\": \"gbdt\",\n         \"feature_fraction\": 0.9,\n         \"bagging_freq\": 1,\n         \"bagging_fraction\": 0.9,\n         \"bagging_seed\": 11,\n         \"metric\": 'rmse',\n         \"lambda_l1\": 0.1,\n         \"verbosity\": -1,\n         \"nthread\": 4,\n         \"random_state\": 4950}\n\n\nclf=lgb.train(params,d_train,100)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e2fe9be5fd1c4b249d2e161c13f2bf6b0e21227"},"cell_type":"code","source":"test.drop([\"first_active_month\",\"card_id\",\"hist_purchase_date_max\", \"hist_purchase_date_min\"],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9e96be3629ac56e050770335dfbe7e2f46cd082"},"cell_type":"code","source":"\ny_pred=clf.predict(test)\ny_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ae121f94ab4e332dedef28aa8d86721261d22b3"},"cell_type":"code","source":"sample_sub=pd.read_csv('../input/sample_submission.csv')\nsample_sub['target']=y_pred\nsample_sub.to_csv(\"submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f21a61d0c942fea13bc8b66aa6a2a2767edc9dcd"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}