{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"LSTM example","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# lstm for time series forecasting\nfrom numpy import sqrt\nfrom numpy import asarray\nfrom pandas import read_csv\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.layers import LSTM\n\n\n# split a univariate sequence into samples\ndef split_sequence(sequence, n_steps):\n    X, y = list(), list()\n    for i in range(len(sequence)):\n        # find the end of this pattern\n        end_ix = i + n_steps\n        # check if we are beyond the sequence\n        if end_ix > len(sequence) - 1:\n            break\n        # gather input and output parts of the pattern\n        seq_x, seq_y = sequence[i:end_ix], sequence[end_ix]\n        X.append(seq_x)\n        y.append(seq_y)\n    return asarray(X), asarray(y)\n\n\n# load the dataset\npath = 'https://raw.githubusercontent.com/jbrownlee/Datasets/master/monthly-car-sales.csv'\ndf = read_csv(path, header=0, index_col=0, squeeze=True)\n# retrieve the values\nvalues = df.values.astype('float32')\n# specify the window size\nn_steps = 5\n# split into samples\nX, y = split_sequence(values, n_steps)\n# reshape into [samples, timesteps, features]\nX = X.reshape((X.shape[0], X.shape[1], 1))\n# split into train/test\nn_test = 12\nX_train, X_test, y_train, y_test = X[:-n_test], X[-n_test:], y[:-n_test], y[-n_test:]\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)\n# define model\nmodel = Sequential()\nmodel.add(LSTM(100, activation='relu', kernel_initializer='he_normal', input_shape=(n_steps, 1)))\nmodel.add(Dense(50, activation='relu', kernel_initializer='he_normal'))\nmodel.add(Dense(50, activation='relu', kernel_initializer='he_normal'))\nmodel.add(Dense(1))\n# compile the model\nmodel.compile(optimizer='adam', loss='mse', metrics=['mae'])\n# fit the model\nmodel.fit(X_train, y_train, epochs=350, batch_size=32, verbose=2, validation_data=(X_test, y_test))\n# evaluate the model\nmse, mae = model.evaluate(X_test, y_test, verbose=0)\nprint('MSE: %.3f, RMSE: %.3f, MAE: %.3f' % (mse, sqrt(mse), mae))\n# make a prediction\nrow = asarray([18024.0, 16722.0, 14385.0, 21342.0, 17180.0]).reshape((1, n_steps, 1))\nyhat = model.predict(row)\nprint('Predicted: %.3f' % (yhat))","metadata":{"execution":{"iopub.status.busy":"2022-08-20T18:57:49.821596Z","iopub.execute_input":"2022-08-20T18:57:49.822553Z","iopub.status.idle":"2022-08-20T18:58:16.998589Z","shell.execute_reply.started":"2022-08-20T18:57:49.822453Z","shell.execute_reply":"2022-08-20T18:58:16.997731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Amex competition partial data with LSTM, based on C:\\John\\git\\vas\\kaggle\\americanExpress\\amex2.py","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom pyarrow.parquet import ParquetFile\nimport pyarrow as pa\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:30:45.849244Z","iopub.execute_input":"2022-08-20T21:30:45.849786Z","iopub.status.idle":"2022-08-20T21:30:45.880458Z","shell.execute_reply.started":"2022-08-20T21:30:45.849658Z","shell.execute_reply":"2022-08-20T21:30:45.879510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_DIR = r\"../input/amex-default-prediction/\"\ndef get_nrows_train_with_label(nrows):\n    if False:\n        pf = ParquetFile(_DIR + 'test.parquet')\n        first_ten_rows = next(pf.iter_batches(batch_size = nrows))\n        df = pa.Table.from_batches([first_ten_rows]).to_pandas()\n    df = pd.read_csv(_DIR + 'train_data.csv', nrows=nrows)\n    # print(df.head())\n\n    targets = pd.read_csv(_DIR + 'train_labels.csv')\n\n    # pass\n\n    train = df.merge(targets, on='customer_ID')#, how='left')\n# train.target = train.target.astype('int8')\n\n    # print(train.head())\n    return train\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:33:09.809882Z","iopub.execute_input":"2022-08-20T21:33:09.810301Z","iopub.status.idle":"2022-08-20T21:33:09.817566Z","shell.execute_reply.started":"2022-08-20T21:33:09.810269Z","shell.execute_reply":"2022-08-20T21:33:09.816419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_train_with_label = get_nrows_train_with_label(100000)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:33:24.749466Z","iopub.execute_input":"2022-08-20T21:33:24.749854Z","iopub.status.idle":"2022-08-20T21:33:33.490004Z","shell.execute_reply.started":"2022-08-20T21:33:24.749822Z","shell.execute_reply":"2022-08-20T21:33:33.488933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Let's use D_39 as the variable to predict label using LSTM\n\n\n#for simplicity let's keep only customers with 13 rows\n# df_train_with_label.groupby('customer_ID').count()['target'].reset_index().groupby('target').count()['customer_ID']\ncustomers_with_13_rows = df_train_with_label.groupby('customer_ID').count()['target'] == 13\ncustomers_with_13_rows = customers_with_13_rows[customers_with_13_rows]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:33:45.038716Z","iopub.execute_input":"2022-08-20T21:33:45.039294Z","iopub.status.idle":"2022-08-20T21:33:45.344672Z","shell.execute_reply.started":"2022-08-20T21:33:45.039244Z","shell.execute_reply":"2022-08-20T21:33:45.343093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import LSTM\nfrom tensorflow.keras.layers import Dense\n\ndf_train_with_label2 = df_train_with_label.merge(customers_with_13_rows, left_on='customer_ID', right_index=True, suffixes=('', '_y')).drop('target_y',axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:53:18.115028Z","iopub.execute_input":"2022-08-20T21:53:18.115417Z","iopub.status.idle":"2022-08-20T21:53:18.762667Z","shell.execute_reply.started":"2022-08-20T21:53:18.115386Z","shell.execute_reply":"2022-08-20T21:53:18.761354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nn_steps = 13\n\ndf_train_with_label2.S_2 = pd.to_datetime( df_train_with_label2.S_2 )\n\ndf_train_with_label2.sort_values(['customer_ID', 'S_2'])\n\ndf_train_with_label2.set_index(['customer_ID', 'S_2'], inplace=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:53:21.975126Z","iopub.execute_input":"2022-08-20T21:53:21.975520Z","iopub.status.idle":"2022-08-20T21:53:22.142611Z","shell.execute_reply.started":"2022-08-20T21:53:21.975486Z","shell.execute_reply":"2022-08-20T21:53:22.141450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.pauldesalvo.com/how-to-apply-a-forward-fill-ffill-to-groups-in-pandas/\nFEATURES = ['D_39','D_47']\n\nfor c in FEATURES:\n    print(c)\n    df_train_with_label2[c] = df_train_with_label2.groupby('customer_ID')[c].transform(lambda x: x.ffill())\n    df_train_with_label2[c] = df_train_with_label2.groupby('customer_ID')[c].transform(lambda x: x.bfill())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:58:00.712361Z","iopub.execute_input":"2022-08-20T21:58:00.713101Z","iopub.status.idle":"2022-08-20T21:58:11.951038Z","shell.execute_reply.started":"2022-08-20T21:58:00.713061Z","shell.execute_reply":"2022-08-20T21:58:11.949950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ffill then bfill for same customerID\n#very slow below compared to above\nif False:\n    for id, new_df in df_train_with_label2.groupby ('customer_ID'):\n        df_train_with_label2.loc[id,:] = df_train_with_label2.loc[id,:].fillna(method='ffill')\n        df_train_with_label2.loc[id,:] =   df_train_with_label2.loc[id,:].fillna(method='bfill')\n    # break\nprint('done')\n# y.shape\ndf_train_with_label2.max()\n\nmax_rows = pd.get_option('display.max_rows')\npd.set_option('display.max_rows', None)\nprint(df_train_with_label2.isnull().sum().sort_index())\npd.set_option('display.max_rows', max_rows)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T21:58:44.267201Z","iopub.execute_input":"2022-08-20T21:58:44.268108Z","iopub.status.idle":"2022-08-20T21:58:44.455435Z","shell.execute_reply.started":"2022-08-20T21:58:44.268058Z","shell.execute_reply":"2022-08-20T21:58:44.454281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\n\nweight0 = df_train_with_label2.groupby('target').count()['D_39']\nweight1 = weight0/weight0.sum()\n\n\n# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\nimport numpy as np\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\n#split customer_ID into test amd train\nids = df_train_with_label2.reset_index()['customer_ID'].unique()\nId_train, Id_test, y_train, y_test = train_test_split(ids, ids, test_size = 0.3, random_state = 0)\n\nsample_weight = np.ones(shape=(len(y_train),))\nsample_weight[y_train == 1] = weight1[1]\nsample_weight[y_train == 0] = weight1[0]\n\n\npd.DataFrame(Id_train, columns=['customer_ID'])\ndf_train_with_label_train = df_train_with_label2.reset_index().merge(pd.DataFrame(Id_train, columns=['customer_ID']), on='customer_ID').set_index(['customer_ID', 'S_2'])#, how='left')\ndf_train_with_label_test = df_train_with_label2.reset_index().merge(pd.DataFrame(Id_test, columns=['customer_ID']), on='customer_ID').set_index(['customer_ID', 'S_2'])#, how='left')\n\ny = df_train_with_label_train.groupby('customer_ID').mean()['target'].to_numpy()\ny_test = df_train_with_label_test.groupby('customer_ID').mean()['target'].to_numpy()\n\n\nFEATURES = ['D_39','D_47']\n# FEATURES = ['D_47']\nX= df_train_with_label_train[FEATURES].to_numpy().reshape(-1,n_steps,len(FEATURES))\nX_test= df_train_with_label_test[FEATURES].to_numpy().reshape(-1,n_steps,len(FEATURES))\n\nmodel = Sequential()\nmodel.add(LSTM(100, activation='relu', kernel_initializer='he_normal', input_shape=(n_steps, len(FEATURES))))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam')#, metrics=['accuracy'])\nhistory = model.fit(X, y, epochs=10, verbose=1, sample_weight=sample_weight)\nyhat=np.round((model.predict(X_test)>=0.5)).astype(int).reshape(-1)\nprint(amex_metric_mod(y_test, yhat))\n\nmax_rows = pd.get_option('display.max_rows')\npd.set_option('display.max_rows', None)\nprint(df_train_with_label2.isnull().sum().sort_index())\npd.set_option('display.max_rows', max_rows)\n\n# plot learning curves\nfrom matplotlib import pyplot\npyplot.title('Learning Curves')\npyplot.xlabel('Epoch')\npyplot.ylabel('Cross Entropy')\npyplot.plot(history.history['loss'], label='train')\n# pyplot.plot(history.history['val_loss'], label='val')\npyplot.legend()\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:02:38.121826Z","iopub.execute_input":"2022-08-20T22:02:38.122271Z","iopub.status.idle":"2022-08-20T22:02:58.329806Z","shell.execute_reply.started":"2022-08-20T22:02:38.122230Z","shell.execute_reply":"2022-08-20T22:02:58.328913Z"},"trusted":true},"execution_count":null,"outputs":[]}]}