{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\n#References:\n#https://www.kaggle.com/yuliagm/talkingdata-eda-plus-time-patterns\n#https://arxiv.org/abs/1604.06737\n#https://github.com/fastai/fastai/blob/master/courses/dl1/lesson3-rossman.ipynb\n#Many past and present kaggle kernels\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordbatch.extractors import WordSeq\nimport wordbatch\n\n%matplotlib inline\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os,gc\nprint(os.listdir(\"../input\"))\nfrom sklearn.model_selection import train_test_split\nfrom keras.layers import Embedding, Input, Dense, concatenate, Flatten, Conv1D\nfrom keras.models import Model\nfrom keras.preprocessing.text import Tokenizer, text_to_word_sequence\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"collapsed":true},"cell_type":"code","source":"traindf = pd.read_csv('../input/train_sample.csv')\n# traindf_sample = pd.read_csv(\"../input/train.csv\", skiprows=160000000, nrows=40000000,\n#                              low_memory=False,header=None)\n# traindf_sample.columns = traindf.columns\ntestdf = pd.read_csv('../input/test.csv')","execution_count":2,"outputs":[]},{"metadata":{"_uuid":"607a218e9962e13d1b4b2af883e5073ef29b54c7","_cell_guid":"f9d060db-3174-48f5-971a-ad074d8059fc","trusted":true},"cell_type":"code","source":"print(traindf.loc[~pd.isnull(traindf['attributed_time']),:].shape[0]/traindf.shape[0])\n# print(traindf_sample.loc[~pd.isnull(traindf_sample['attributed_time']),:].shape[0]/traindf_sample.shape[0])","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"94b4c6b7f1c05b709df9ffea8f60b17079561d5b","_cell_guid":"8dce7d66-39ce-4a20-b723-545de884ed80","collapsed":true,"trusted":true},"cell_type":"code","source":"traindf['click_time'] = pd.to_datetime(traindf['click_time'])","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"7cb6d00a1703cfa17db2ab77e66824aa60f728fc","_cell_guid":"b7cc7375-106b-46c0-8683-436155ec5b1b","trusted":true},"cell_type":"code","source":"# traindf[['app','device','os','channel']].corr()\n# traindf[['ip','channel']].corr()\ntraindf['click_time'].dtypes","execution_count":5,"outputs":[]},{"metadata":{"_uuid":"85ca978b28a04bef87aeeeed79eb4415946c194c","_cell_guid":"cc77c465-0ebe-4faf-9b2d-15b3aa5cac83","collapsed":true,"trusted":true},"cell_type":"code","source":"def create_features(traindf):\n    if traindf['click_time'].dtypes != '<M8[ns]':\n        traindf['click_time'] = pd.to_datetime(traindf['click_time'])\n    features_text = 'XI'+traindf['ip'].astype(str)+' XA'+traindf['app'].astype(str)+' XO'+traindf['os'].astype(str) + \" XC\" + traindf['channel'].astype(str) + \\\n    \" XH\" + (traindf['click_time'].dt.hour).astype(str) + ' XD'+traindf['device'].astype(str) + \\\n    ' XDXO'+\"_\"+traindf['device'].astype(str)+\"_\"+traindf['os'].astype(str) + \\\n    \" XAXOXC\" +\"_\"+traindf['app'].astype(str)+\"_\"+traindf['os'].astype(str) +\"_\"+traindf['channel'].astype(str)\n    return features_text.values","execution_count":6,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"dce0285f69cedcc1e9de8822b0454e521929ffba"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"42605838b9c639547ae499476e8af7b927c74f43","_cell_guid":"61597787-d464-4e9c-bef5-cb7eb21c71c1","collapsed":true,"trusted":true},"cell_type":"code","source":"def get_deltas(data):\n    clk_time = pd.to_datetime(data['click_time'])\n    att_time = pd.to_datetime(data['attributed_time'],errors='coerce')\n    data['click_time'] = clk_time\n    data['attributed_time'] = att_time\n    deltas = (att_time-clk_time).dt.seconds\n    deltas = deltas.replace(to_replace=np.NaN,value=0.0)\n    data['deltas'] = deltas\n    data['reg_deltas']  = (deltas/deltas.max()).values.astype(np.float32)\n    return data\n\ndef create_features(traindf):\n    if traindf['click_time'].dtypes != '<M8[ns]':\n        traindf['click_time'] = pd.to_datetime(traindf['click_time'])\n    features_text = 'XA'+traindf['app'].astype(str)+' XO'+traindf['os'].astype(str) + \" XC\" + traindf['channel'].astype(str) + \\\n    \" XH\" + (traindf['click_time'].dt.hour).astype(str) + ' XD'+traindf['device'].astype(str) + \\\n    ' XDXO'+\"_\"+traindf['device'].astype(str)+\"_\"+traindf['os'].astype(str) + \\\n    \" XAXOXC\" +\"_\"+traindf['app'].astype(str)+\"_\"+traindf['os'].astype(str) +\"_\"+traindf['channel'].astype(str)\n    return features_text.values","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"fe254bd636e8cb479ca7bc5714adbb9604f13721","_cell_guid":"5fe17768-a249-489b-970d-2a26368b40c0","trusted":true,"collapsed":true},"cell_type":"code","source":"features_text = create_features(traindf)","execution_count":8,"outputs":[]},{"metadata":{"_uuid":"9f15d73d1f391ab659b35b09c9c20ef867d4ea05","_cell_guid":"524a0fa8-d323-441b-a5cb-834c730cb2aa","trusted":true},"cell_type":"code","source":"# features_text\n# test_features\n\n# features_text\nfeatures_text","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"683e389f26de0d0bfcbcd7d62dcd59acaf5fbb0e","_cell_guid":"36572ed2-fbcb-4a41-aa98-63036917377b","trusted":false,"collapsed":true},"cell_type":"code","source":"# features_text = features_text.tolist()\n# features_text = create_features(traindf_sample)\ntest_features","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a9fc0e6c7c5ba5a7ce39e686f675014f1ac618e3","_cell_guid":"d8d613de-66f3-407c-9a26-f84f63d58887","trusted":false,"collapsed":true},"cell_type":"code","source":"# ft = create_features(traindf)\ntest_features = create_features(testdf)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a5cc6ca81058c5059dfb5e3637c3ef6f4d44d5b","_cell_guid":"ede85ce3-755c-4127-a759-ad2e1e6e8667","trusted":false,"collapsed":true},"cell_type":"code","source":"# tokenizer = Tokenizer()\n# train_t = tokenizer.fit_on_texts(features_text)\n# train_t = tokenizer.texts_to_sequences(features_text)\n\n# tokenizer = Tokenizer()\ntest_t = tokenizer.fit_on_texts(test_features)\ntest_t = tokenizer.texts_to_sequences(test_features)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ad7cfab5c535279ba99689df0e68a62a6d7bccd","_cell_guid":"86b53238-bd4c-44f3-bb77-6cb7636e0295","trusted":false,"collapsed":true},"cell_type":"code","source":"test_features.shape\n# np.array(test_t).shape\n# np.array(train_t).shape\n# train_t\n# len(test_t[0])\n# np.array(test_t)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a94cd25d3367cd1609942e423e46cdb1971552b1","_cell_guid":"d9b2dabf-4105-4c52-a051-adb2eba92edc","trusted":false,"collapsed":true},"cell_type":"code","source":"# np.array(train_t)\n# test_array  = np.hstack([np.array(t) for t in test_t])\nnp.array(test_t).shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b5c640da2184e157e219634188719792f91b329d","_cell_guid":"a6c5a38a-806e-4d34-bb60-05d562f5f3b7","collapsed":true,"trusted":false},"cell_type":"code","source":"test_t = tokenizer.fit_on_texts(test_features)\ntest_t = tokenizer.texts_to_sequences(test_features)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"979ccb124eab6c0a4febe2acdc262ad0754a58b0","_cell_guid":"48857a08-dee6-43d8-a395-09aa5f313904","trusted":false,"collapsed":true},"cell_type":"code","source":"# train_t\nword_index = tokenizer.word_index\nlen(word_index)\n# np.hsplit(traindf[cols].values,traindf[cols].shape[1])[0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0dc7cf85e10f033b72e6ba1d4ee902a28a3baf02","_cell_guid":"6c79f254-9d19-445d-ae55-e40ba61d6abe","trusted":false,"collapsed":true},"cell_type":"code","source":"# len(train_t[0]\no = pd.Series([len(l) for l in test_t])\no.min()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"933413f7bcf3de287ff9891bb054b077bd5eb118","_cell_guid":"48202de2-9d01-428c-9f5e-9d82c5073b16","trusted":false,"collapsed":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a1e78c7cd9e286d837d38dd2837168f91e550989","_cell_guid":"c55c448c-7abf-44a5-b455-0ebb6a82b996","collapsed":true,"trusted":false},"cell_type":"code","source":"inp = Input((len(train_t[0]),))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf96b8510db6b66ff884cecceea63bdd52a36c39","_cell_guid":"8631d02f-67d1-4ef4-94fc-2227094d6503","trusted":false,"collapsed":true},"cell_type":"code","source":"x = Embedding(len(word_index)+10000,5)(inp)\nx = Conv1D(32,5)(x)\nx = Flatten()(x)\nis_attributed = Dense(1,activation='sigmoid',name='is_attributed')(x)\ndeltas_regr = Dense(1,activation='sigmoid',name='reg_deltas')(x)\nseq_model = Model(inp,[is_attributed,deltas_regr])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7681cea7137ab1b3b90b26a4190d44df520bd0d","_cell_guid":"51a5aad6-f435-4c81-abe7-a7a47082cd36","trusted":false,"collapsed":true},"cell_type":"code","source":"seq_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c6b1a7d38faa40df26cec50553be19ed0489698e","_cell_guid":"5606d54a-c2cf-4dd9-a0a8-1f6af41927df","trusted":false,"collapsed":true},"cell_type":"code","source":"Y = traindf['is_attributed'].values\nreg_deltas = traindf['reg_deltas'].values","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"_cell_guid":"4d65710b-dd12-4535-b61d-42c4d2c8514b","_uuid":"efb81dc0e1b8a04c9015bda3d5ec2e8e0260692d","trusted":false,"collapsed":true},"cell_type":"code","source":"#Compile and fit model:\nseq_model.compile('adam',['binary_crossentropy','mse'])\nseq_model.fit(np.array(train_t),[Y,reg_deltas],batch_size=512*16,validation_split=0.1,epochs=7)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7dbbecaaffc17e6061a4f4a8020f3b6d784aa497","_cell_guid":"c6444d24-214f-4f62-ab0e-c062210021ea","trusted":false,"collapsed":true},"cell_type":"code","source":"seq_model_test_preds = seq_model.predict(np.array(test_t),verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a35c77dd51b814eec4d01c9e6992eac027e8379","_cell_guid":"200510d6-a0db-4b86-bbd9-f93f7df37ec1","trusted":false,"collapsed":true},"cell_type":"code","source":"# seq_model_test_preds[0]\n#Create submission:\noutdf = pd.DataFrame()\noutdf['click_id'] = testdf['click_id']\noutdf['is_attributed'] = seq_model_test_preds[0]\noutdf.to_csv('sub.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53028407d3ac25c08e0fcbc9ba381e7f7d964e3b","_cell_guid":"0b4da31a-5bed-49dc-af79-5168948f8430","trusted":false,"collapsed":true},"cell_type":"code","source":"import gc; gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"203bee436a59abf6161ce1994b1b89ba043f24a9","_cell_guid":"5945edbc-d033-44c5-8571-f034fc8281d5","trusted":false,"collapsed":true},"cell_type":"code","source":"# traindf_sample = get_deltas(traindf_sample)\ntraindf = get_deltas(traindf)\nclick_counts = traindf['deltas']\nclick_counts.plot()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3c40896bcb5bdfb1c49b8a8eac8d0dd45e55846e","_cell_guid":"742d3a59-c252-4572-a2f4-f66ef0b9401a","collapsed":true,"trusted":false},"cell_type":"code","source":"print('Max Delta datapoint: \\n{}'.format(traindf.loc[np.argmax(traindf['deltas'].values),:]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b234c69a105adba9d9771e43ec64a089f012b04","_cell_guid":"9cc4bc61-968d-45b1-805c-6b5eec99cb27","collapsed":true,"trusted":false},"cell_type":"code","source":"#Hourly click count feature:\ntraindf['click_time'] = pd.to_datetime(traindf['click_time'])\ntraindf_sample['click_time'] = pd.to_datetime(traindf_sample['click_time'])\ntestdf['click_time'] = pd.to_datetime(testdf['click_time'])\ntraindf['hour'] = traindf['click_time'].dt.hour\ntraindf_sample['hour'] = traindf_sample['click_time'].dt.hour\ntestdf['hour'] = testdf['click_time'].dt.hour\ntraindf['click_time'].dt.hour.value_counts().reset_index().plot(kind='scatter',x='index',y='click_time')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"512f4b35c936468c8427456e2456e76c074a2842","_cell_guid":"6b46d865-732f-44a3-a438-2454eb4d7fd3","collapsed":true,"trusted":false},"cell_type":"code","source":"#Values only on four days:\ntraindf['click_time'].dt.dayofweek.value_counts().reset_index().plot(kind='scatter',x='index',y='click_time')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ae75fdb7a0f51e59469a248b72b5c0307ce4a745","_cell_guid":"e7912f58-1511-49c1-ae5a-65f7337f2932","collapsed":true,"trusted":false},"cell_type":"code","source":"traindf['click_time'].dt.second.value_counts().reset_index().plot(kind='scatter',x='index',y='click_time')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"032cee63a67271fb3574bca22a4f126c68f09495","_cell_guid":"67efab1c-5b66-46b6-afd3-6fff75cc1e24","collapsed":true,"trusted":false},"cell_type":"code","source":"cols = ['ip','app','device', 'os', 'channel','hour']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a347af262fe4d7d254d786087e2dc0890352f6b6","_cell_guid":"2157a3e4-0d4f-4f0b-be6f-308c8adea557","collapsed":true,"trusted":false},"cell_type":"code","source":"def split_t(X):\n    return np.hsplit(X,X.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ccf75de53148e0a3aca81a93fb011d2cc1a91854","_cell_guid":"a7d3864f-cb0b-42d3-bc28-e087dc9de33a","collapsed":true,"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bbfb959b0f3be5f48b34d5acee4c6b9d181cc670","_cell_guid":"ffac9fe0-1af8-4c51-b8d5-7bc8cde0d799","collapsed":true,"trusted":false},"cell_type":"code","source":"X = split_t(traindf_sample[cols])\nXtest = split_t(testdf[cols])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"906c6797b7b60c3388ea3f6983870a6185c6905a","_cell_guid":"f70c9711-826a-443a-a5bb-e83dbaf77bcc","collapsed":true,"trusted":false},"cell_type":"code","source":"Y = traindf_sample['is_attributed'].values\nreg_deltas = traindf_sample['reg_deltas'].values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2b0861d62a31e9eb21a6334498d4a3fd562dc542","_cell_guid":"78149b17-a670-48a8-9516-0c515975ef94","collapsed":true,"trusted":false},"cell_type":"code","source":"def make_emb(data,col):\n    col_dim = data[col].unique().shape[0]\n    if col_dim>1000:\n        out_dim = 5\n        col_dim = 600000\n    elif col_dim<1000 and col_dim>100:\n        out_dim = 5\n        col_dim = 10000\n    elif col_dim<100 and col_dim>10:\n        out_dim = 5\n        col_dim = 5000\n    else:\n        out_dim = 5\n    emb = Embedding(col_dim,out_dim,input_length=1,embeddings_initializer='glorot_normal',name=col+'_emb')\n    inp = Input((1,),name=col+'_input')\n    return inp,emb(inp)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8ac7307eec9ef9be67244fe95b6f2964e254a466","_cell_guid":"5a6166cd-9ebd-410a-a65d-2f42c4cc3ec1","collapsed":true,"trusted":false},"cell_type":"code","source":"embs = [make_emb(traindf,col) for col in cols]\ninps = [inp for inp,emb in embs]\nx = concatenate([emb for inp,emb in embs])\nx = Flatten()(x)\nx = Dense(20,activation='relu')(x)\nx = Dense(10,activation='relu')(x)\nis_attributed = Dense(1,activation='sigmoid',name='is_attributed')(x)\ndeltas_regr = Dense(1,activation='sigmoid',name='reg_deltas')(x)\nmodel = Model(inps,[is_attributed,deltas_regr])","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"_cell_guid":"2dd4c595-78f9-444a-8d03-fcb2c9ee6721","_uuid":"ecc519c7a462fa85f535731d48ddae80a5473d31","collapsed":true,"trusted":false},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"_cell_guid":"85eb8318-ba42-4d5b-bd6c-08c99eb22fbf","_uuid":"282b94bfa4ec3696c0f32d153ab6ab392d8101b0","collapsed":true,"trusted":false},"cell_type":"code","source":"#Compile and fit model:\nmodel.compile('adam',['binary_crossentropy','mse'])\nmodel.fit(X,[Y,reg_deltas],batch_size=512*16,validation_split=0.1,epochs=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d27d27f14d95f1bc5054640141b861d72f36d96a","_cell_guid":"7e26b8e4-3e75-4a13-845e-85432a19a1ac","collapsed":true,"trusted":false},"cell_type":"code","source":"test_preds = model.predict(Xtest)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb8b0f531a6a24b638109697ec60e29609afb643","_cell_guid":"f7bf1458-a049-421e-9d10-c2361c375680","collapsed":true,"trusted":false},"cell_type":"code","source":"#Create submission:\noutdf = pd.DataFrame()\noutdf['click_id'] = testdf['click_id']\noutdf['is_attributed'] = seq_model_test_preds[0]\noutdf.to_csv('sub.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45b064681a6eac34633c1a01e31ca62253b479eb","_cell_guid":"67ac2de0-3457-4860-ab4d-63a483d2e90e","collapsed":true,"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}