{"cells":[{"metadata":{"collapsed":true,"_uuid":"df18f7161670481a0dba44d75aee1aa59ff6352d","trusted":false},"cell_type":"code","source":"\"\"\"\nBased on https://www.kaggle.com/georsara1/95-auc-score-in-train-sample-with-neural-nets/code\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":false},"cell_type":"code","source":"%matplotlib inline\n\nfrom datetime import datetime\nimport time\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom keras.callbacks import TensorBoard\nfrom keras.models import Sequential\nfrom keras.layers import Activation, Dense, Dropout\nfrom keras import optimizers\nfrom sklearn.metrics import confusion_matrix,accuracy_score, roc_curve, auc\nsns.set_style(\"whitegrid\")\nnp.random.seed(697)\n\nBATCH_SIZE = 1024\nALL_TRAIN_DATA_SIZE = 184903890\nTRAIN_DATA_SIZE = 1000000\nINPUT_SIZE = 5207","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c0395b602a4dadaa33dfc3706ec6eac51bf2e3dc"},"cell_type":"raw","source":"First preprocess the data to have all the unique categry values.\n\n#!/bin/bash\n\nunzip -p test.csv.zip | cut -d',' -f1 | awk '!x[$0]++' > ip.categories.csv\nunzip -p test.csv.zip | cut -d',' -f2 | awk '!x[$0]++' > app.categories.csv\nunzip -p test.csv.zip | cut -d',' -f3 | awk '!x[$0]++' > device.categories.csv\nunzip -p test.csv.zip | cut -d',' -f4 | awk '!x[$0]++' > os.categories.csv\nunzip -p test.csv.zip | cut -d',' -f5 | awk '!x[$0]++' > channel.categories.csv\n\ncat > time_interval.categories.csv << EOF\ntime_interval\n00\n01\n02\n03\n04\n05\n06\n07\n08\n09\n10\n11\n12\n13\n14\n15\n16\n17\n18\n19\n20\n21\n22\n23\nEOF\n"},{"metadata":{"collapsed":true,"_uuid":"57a4d3d890ad8793a285b795c676f81b30a9ad2d","trusted":false},"cell_type":"code","source":"app = pd.read_csv('app.categories.csv').values.reshape(-1)\ndevice = pd.read_csv('device.categories.csv').values.reshape(-1)\nos = pd.read_csv('os.categories.csv').values.reshape(-1)\nchannel = pd.read_csv('channel.categories.csv').values.reshape(-1)\ntime_interval = pd.read_csv('time_interval.categories.csv').values.reshape(-1)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"4cd48e0fbfd55bbdcc7b43d245048e7f4f30dd3f","trusted":false},"cell_type":"code","source":"tbCallBack = TensorBoard(log_dir='./Graph', histogram_freq=0, write_graph=True, write_images=True)\n\ndef parse_data(df):\n    #---------------------------Pre-processing-------------------------\n    #Create new variables\n    df['time_interval'] = df.click_time.str[11:13]\n\n    #Drop unneeded variables\n    df = df.drop(['ip', 'attributed_time', 'click_time'], axis = 1)\n    \n    #Encode categorical variables to ONE-HOT\n    categorical_columns = ['app', 'device', 'os', 'channel', 'time_interval']\n    \n    df['app'] = df['app'].astype('category', categories=app)\n    df['device'] = df['device'].astype('category', categories=device)\n    df['os'] = df['os'].astype('category', categories=os)\n    df['channel'] = df['channel'].astype('category', categories=channel)\n    df['time_interval'] = df['time_interval'].astype('category', categories=time_interval)\n\n    df = pd.get_dummies(df, columns = categorical_columns) \n    \n    #Get the data ready for the Neural Network\n    train_y = df.is_attributed\n    train_x = df.drop(['is_attributed'], axis = 1)\n    train_x = np.array(train_x)\n    train_y = np.array(train_y)\n    \n    return train_x, train_y\n\n\ndef data_generator(filename):\n    while True:\n        for chunk in pd.read_csv(filename, chunksize=BATCH_SIZE):\n            yield parse_data(chunk)\n\n            \ndef parse_test_data(df):\n    #---------------------------Pre-processing-------------------------\n    #Create new variables\n    df['time_interval'] = df.click_time.str[11:13]\n\n    #Drop unneeded variables\n    df = df.drop(['ip', 'click_time'], axis = 1)\n    \n    #Encode categorical variables to ONE-HOT\n    categorical_columns = ['app', 'device', 'os', 'channel', 'time_interval']\n    \n    df['app'] = df['app'].astype('category', categories=app)\n    df['device'] = df['device'].astype('category', categories=device)\n    df['os'] = df['os'].astype('category', categories=os)\n    df['channel'] = df['channel'].astype('category', categories=channel)\n    df['time_interval'] = df['time_interval'].astype('category', categories=time_interval)\n\n    df = pd.get_dummies(df, columns = categorical_columns) \n    \n    return np.array(df)\n","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"c0dc9717c5b228e881efeadf8433e42370bed5e6","_cell_guid":"3a1db286-4549-4edf-9ce0-071a0c9de68e","trusted":false},"cell_type":"code","source":"#-------------------Build the Neural Network model-------------------\nprint('Building Neural Network model...')\n\nadam = optimizers.adam(lr = 0.005, decay = 0.0000001)\n\nmodel = Sequential()\nmodel.add(Dense(300, input_dim=INPUT_SIZE,\n                kernel_initializer='normal',\n                activation=\"relu\"))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(224, activation=\"tanh\"))\nmodel.add(Dropout(0.4))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss=\"binary_crossentropy\", optimizer=adam)\n\n#class_weight = {0 : 533.273, 1: 1.}\n\nprint(\"Epochs per whole dataset:\" + str(int(ALL_TRAIN_DATA_SIZE / TRAIN_DATA_SIZE)))\nprint(\"Start training - \" + time.strftime('%X %x %Z'))\n\nhistory = model.fit_generator(data_generator('train.csv.zip'), steps_per_epoch=(TRAIN_DATA_SIZE/BATCH_SIZE), validation_data=data_generator('train_sample.csv.zip'), validation_steps=BATCH_SIZE, epochs=int(ALL_TRAIN_DATA_SIZE / TRAIN_DATA_SIZE), callbacks=[tbCallBack])\n#history = model.fit_generator(data_generator('train_sample.csv.zip'), steps_per_epoch=(100000/BATCH_SIZE), validation_data=data_generator('train.csv.zip'), validation_steps=BATCH_SIZE, epochs=30, callbacks=[tbCallBack])\n      \nprint(\"Finished training - \" + time.strftime('%X %x %Z'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"74333d282410413309bbf5592572b7dc950c58b3","_cell_guid":"e7e97866-c345-4060-b75a-103493fc5c31","trusted":false},"cell_type":"code","source":"# summarize history for loss\nplt.figure()\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper right')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"e535c0437a9a21ca1b69f0ad93f1e6bcac1ff64d","trusted":false},"cell_type":"code","source":"#Save model\n\n# serialize model to JSON\nmodel_prefix = datetime.now().strftime(\"%Y%m%d-%H:%m:%S\")\nmodel_json = model.to_json()\nwith open(model_prefix + \"_model.json\", \"w\") as json_file:\n    json_file.write(model_json)\n    \n# serialize weights to HDF5\nmodel.save_weights(model_prefix + \"_model.h5\")\nprint(\"Saved model to disk\")","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"72344bd95efdc5d9870d3a101e178cfe478f5f95","trusted":false},"cell_type":"code","source":"#predict\nprint(\"Start predicting - \" + time.strftime('%X %x %Z'))\n\nwith_header = True\nfor chunk in pd.read_csv(\"test.csv.zip\", chunksize=1000):\n    sub = pd.DataFrame()\n    sub['click_id'] = chunk['click_id']\n    chunk.drop('click_id', axis = 1, inplace=True)\n    x = parse_test_data(chunk)\n    sub['is_attributed'] = model.predict_on_batch(x)\n    \n    if with_header:\n        sub.to_csv('out_keras_try.csv', header=True, index=False, mode='w')\n        with_header = False\n    else:\n        sub.to_csv('out_keras_try.csv', header=False, index=False, mode='a')\n        \nprint(\"Finished predicting - \" + time.strftime('%X %x %Z'))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"d0b92205c82d3c2268955d0d2fb1e8c0df2bc83c","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}