{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         pass\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-16T14:32:11.726598Z","iopub.execute_input":"2022-03-16T14:32:11.726843Z","iopub.status.idle":"2022-03-16T14:32:11.751449Z","shell.execute_reply.started":"2022-03-16T14:32:11.726771Z","shell.execute_reply":"2022-03-16T14:32:11.750847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport cudf\nimport cuml\nimport cupy\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:32:11.752627Z","iopub.execute_input":"2022-03-16T14:32:11.754613Z","iopub.status.idle":"2022-03-16T14:32:19.944676Z","shell.execute_reply.started":"2022-03-16T14:32:11.754574Z","shell.execute_reply":"2022-03-16T14:32:19.943927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = '../input/h-and-m-personalized-fashion-recommendations/'\nimage_path = '../input/h-and-m-personalized-fashion-recommendations/images/'","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:32:19.945926Z","iopub.execute_input":"2022-03-16T14:32:19.946182Z","iopub.status.idle":"2022-03-16T14:32:19.950961Z","shell.execute_reply.started":"2022-03-16T14:32:19.946149Z","shell.execute_reply":"2022-03-16T14:32:19.950035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(data_path+'transactions_train.csv', dtype={'article_id': str, 't_dat': object, 'customer_id': str, 'price': float, 'sales_channel_id': int})\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:32:19.953005Z","iopub.execute_input":"2022-03-16T14:32:19.95346Z","iopub.status.idle":"2022-03-16T14:33:23.263838Z","shell.execute_reply.started":"2022-03-16T14:32:19.953352Z","shell.execute_reply":"2022-03-16T14:33:23.263062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" - Pre-processing","metadata":{}},{"cell_type":"code","source":"df['t_dat'] = pd.to_datetime(df['t_dat'])\ndf['timestamp'] = pd.to_datetime(df['t_dat'], unit='ms').astype(np.int64)//1e9\ndf['customer_id'] = df['customer_id'].astype('str')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:33:55.943627Z","iopub.execute_input":"2022-03-16T14:33:55.944427Z","iopub.status.idle":"2022-03-16T14:34:06.01438Z","shell.execute_reply.started":"2022-03-16T14:33:55.944374Z","shell.execute_reply":"2022-03-16T14:34:06.013553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:34:22.784404Z","iopub.execute_input":"2022-03-16T14:34:22.784981Z","iopub.status.idle":"2022-03-16T14:34:24.567996Z","shell.execute_reply.started":"2022-03-16T14:34:22.784939Z","shell.execute_reply":"2022-03-16T14:34:24.567314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_cnt = df.isna().sum()\nnull_cnt","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:34:25.609204Z","iopub.execute_input":"2022-03-16T14:34:25.609712Z","iopub.status.idle":"2022-03-16T14:34:31.894069Z","shell.execute_reply.started":"2022-03-16T14:34:25.609673Z","shell.execute_reply":"2022-03-16T14:34:31.893345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Unique items in each field\ntotal_transactions = len(df)\nfor col in ['customer_id', 'article_id']:\n    uq = len(df[col].unique())\n    print('Unique in', col, uq, 'Density:', total_transactions/uq)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:34:32.800207Z","iopub.execute_input":"2022-03-16T14:34:32.800461Z","iopub.status.idle":"2022-03-16T14:34:42.880815Z","shell.execute_reply.started":"2022-03-16T14:34:32.800432Z","shell.execute_reply":"2022-03-16T14:34:42.880049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uniq_customer = list(df['customer_id'].unique())\nuniq_article = list(df['article_id'].unique())\nprint(len(uniq_customer), len(uniq_article))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:34:46.317498Z","iopub.execute_input":"2022-03-16T14:34:46.317814Z","iopub.status.idle":"2022-03-16T14:34:56.284842Z","shell.execute_reply.started":"2022-03-16T14:34:46.317779Z","shell.execute_reply":"2022-03-16T14:34:56.284051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uniq_customer[:3]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:35:45.001237Z","iopub.execute_input":"2022-03-16T14:35:45.00177Z","iopub.status.idle":"2022-03-16T14:35:45.006551Z","shell.execute_reply.started":"2022-03-16T14:35:45.00173Z","shell.execute_reply":"2022-03-16T14:35:45.005875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = df.sample(90_000, random_state=26)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:35:46.164004Z","iopub.execute_input":"2022-03-16T14:35:46.164669Z","iopub.status.idle":"2022-03-16T14:35:47.621155Z","shell.execute_reply.started":"2022-03-16T14:35:46.164629Z","shell.execute_reply":"2022-03-16T14:35:47.620378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itemSeq = sub_df.groupby('customer_id')['article_id'].apply(lambda x: x.to_string(index=False, header=False))\nitemSeq","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:35:48.513524Z","iopub.execute_input":"2022-03-16T14:35:48.514227Z","iopub.status.idle":"2022-03-16T14:36:10.48728Z","shell.execute_reply.started":"2022-03-16T14:35:48.514191Z","shell.execute_reply":"2022-03-16T14:36:10.486352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"orderedItems = []\ncustomers = []\nitems = []\nfor key, seq in itemSeq.iteritems():\n    s = seq.split('\\n')\n#     print(key)\n    customers.append(key)\n    items.append(' '.join(s))\n    if len(s) > 1:\n        orderedItems.append(' '.join(s))\norderedItems","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:11.601754Z","iopub.execute_input":"2022-03-16T14:36:11.602155Z","iopub.status.idle":"2022-03-16T14:36:11.716824Z","shell.execute_reply.started":"2022-03-16T14:36:11.602119Z","shell.execute_reply":"2022-03-16T14:36:11.716007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = pd.DataFrame({'customer_id':customers, 'order_seq': items})\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:12.388738Z","iopub.execute_input":"2022-03-16T14:36:12.389002Z","iopub.status.idle":"2022-03-16T14:36:12.429155Z","shell.execute_reply.started":"2022-03-16T14:36:12.388973Z","shell.execute_reply":"2022-03-16T14:36:12.428489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = tf.keras.preprocessing.text.Tokenizer()\ntokenizer.fit_on_texts(orderedItems)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:17.294741Z","iopub.execute_input":"2022-03-16T14:36:17.295554Z","iopub.status.idle":"2022-03-16T14:36:17.390902Z","shell.execute_reply.started":"2022-03-16T14:36:17.295505Z","shell.execute_reply":"2022-03-16T14:36:17.390178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_items = len(tokenizer.word_index) + 1\ntotal_items","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:18.33973Z","iopub.execute_input":"2022-03-16T14:36:18.339992Z","iopub.status.idle":"2022-03-16T14:36:18.345706Z","shell.execute_reply.started":"2022-03-16T14:36:18.339963Z","shell.execute_reply":"2022-03-16T14:36:18.344872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_sequences = []\nfor line in orderedItems:\n    token_list = tokenizer.texts_to_sequences([line])[0]\n    for i in range(1, len(token_list)):\n        n_gram_sequence = token_list[:i+1]\n        input_sequences.append(n_gram_sequence)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:19.103725Z","iopub.execute_input":"2022-03-16T14:36:19.104205Z","iopub.status.idle":"2022-03-16T14:36:19.190501Z","shell.execute_reply.started":"2022-03-16T14:36:19.104166Z","shell.execute_reply":"2022-03-16T14:36:19.189836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_seq_len = max([len(inp) for inp in input_sequences])\nmax_seq_len","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:20.112673Z","iopub.execute_input":"2022-03-16T14:36:20.113249Z","iopub.status.idle":"2022-03-16T14:36:20.119279Z","shell.execute_reply.started":"2022-03-16T14:36:20.113205Z","shell.execute_reply":"2022-03-16T14:36:20.118516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq = tf.keras.preprocessing.sequence.pad_sequences(input_sequences, maxlen=max_seq_len-1, padding='pre')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:20.848835Z","iopub.execute_input":"2022-03-16T14:36:20.849356Z","iopub.status.idle":"2022-03-16T14:36:20.918559Z","shell.execute_reply.started":"2022-03-16T14:36:20.849312Z","shell.execute_reply":"2022-03-16T14:36:20.917731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:21.607374Z","iopub.execute_input":"2022-03-16T14:36:21.607952Z","iopub.status.idle":"2022-03-16T14:36:21.618377Z","shell.execute_reply.started":"2022-03-16T14:36:21.607911Z","shell.execute_reply":"2022-03-16T14:36:21.61757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = seq[:,:-1]\nlabels = seq[:,-1]\n# One-hot encode with keras convert list to a categorical. The number of classes which is my number of words.\nY = tf.keras.utils.to_categorical(labels, num_classes=total_items)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:22.730878Z","iopub.execute_input":"2022-03-16T14:36:22.731402Z","iopub.status.idle":"2022-03-16T14:36:22.756618Z","shell.execute_reply.started":"2022-03-16T14:36:22.731363Z","shell.execute_reply":"2022-03-16T14:36:22.755902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential()\nmodel.add(tf.keras.layers.Embedding(total_items, 100, input_length=max_seq_len-1))\nmodel.add(tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(150)))\nmodel.add(tf.keras.layers.Dense(total_items, activation='softmax'))\nadam = tf.keras.optimizers.Adam()\nmodel.compile(loss='categorical_crossentropy', optimizer=adam, metrics=['accuracy'])\nhistory = model.fit(X, Y, epochs=25, verbose=1)\nprint(model)\ntf.keras.utils.plot_model(model)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:36:23.57365Z","iopub.execute_input":"2022-03-16T14:36:23.574168Z","iopub.status.idle":"2022-03-16T14:37:43.188711Z","shell.execute_reply.started":"2022-03-16T14:36:23.57413Z","shell.execute_reply":"2022-03-16T14:37:43.187943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def findNext12(seed_text):\n    next_items = 12\n    #Prediction\n#     print(seed_text)\n    ans = \"\"\n    for _ in range(next_items):\n        token_list = tokenizer.texts_to_sequences([seed_text])[0]\n        token_list = tf.keras.preprocessing.sequence.pad_sequences([token_list], maxlen=max_seq_len-1, padding='pre')\n        predicted = np.argmax(model.predict(token_list, verbose=0)[0])\n#         print(predicted)\n        output_word = \"\"\n        for word, index in tokenizer.word_index.items():\n            if index == predicted:\n                output_word = word\n                break\n        seed_text += \" \" + output_word\n        ans += \" \" + output_word\n    return ans[1:]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:38:30.550269Z","iopub.execute_input":"2022-03-16T14:38:30.550583Z","iopub.status.idle":"2022-03-16T14:38:30.560498Z","shell.execute_reply.started":"2022-03-16T14:38:30.550546Z","shell.execute_reply":"2022-03-16T14:38:30.559779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# findNext12('0589544003 0714828001')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:38:31.271673Z","iopub.execute_input":"2022-03-16T14:38:31.272298Z","iopub.status.idle":"2022-03-16T14:38:32.263967Z","shell.execute_reply.started":"2022-03-16T14:38:31.27226Z","shell.execute_reply":"2022-03-16T14:38:32.263291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df['prediction'] = final_df['order_seq'].apply(lambda x: findNext12(x))\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:38:32.679419Z","iopub.execute_input":"2022-03-16T14:38:32.679811Z","iopub.status.idle":"2022-03-16T14:38:53.840041Z","shell.execute_reply.started":"2022-03-16T14:38:32.679777Z","shell.execute_reply":"2022-03-16T14:38:53.839325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = final_df[['customer_id', 'prediction']]\nsubmission[:5]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:39:28.447616Z","iopub.execute_input":"2022-03-16T14:39:28.448176Z","iopub.status.idle":"2022-03-16T14:39:28.5607Z","shell.execute_reply.started":"2022-03-16T14:39:28.448135Z","shell.execute_reply":"2022-03-16T14:39:28.560008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T14:39:50.810627Z","iopub.execute_input":"2022-03-16T14:39:50.811239Z","iopub.status.idle":"2022-03-16T14:39:51.11225Z","shell.execute_reply.started":"2022-03-16T14:39:50.8112Z","shell.execute_reply":"2022-03-16T14:39:51.111516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}