{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport dask.dataframe as dd # for out of memeory processing\nfrom dask.distributed import Client\n\n!pip install pickle5\nimport pickle5 as pickle\nfrom tqdm.notebook import tqdm\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\nimport plotly.express as px\n\nimport dask.array as da","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:14.626724Z","iopub.execute_input":"2022-12-29T11:42:14.627799Z","iopub.status.idle":"2022-12-29T11:42:30.387681Z","shell.execute_reply.started":"2022-12-29T11:42:14.627668Z","shell.execute_reply":"2022-12-29T11:42:30.386472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install polars\n\nimport polars as pl","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:30.390224Z","iopub.execute_input":"2022-12-29T11:42:30.390513Z","iopub.status.idle":"2022-12-29T11:42:41.645797Z","shell.execute_reply.started":"2022-12-29T11:42:30.390483Z","shell.execute_reply":"2022-12-29T11:42:41.644678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the Data\nWe will be using the memory optimized data provided by [Radek](https://www.kaggle.com/code/radek1/howto-full-dataset-as-parquet-csv-files?scriptVersionId=113624955).","metadata":{}},{"cell_type":"code","source":"id2type = pickle.load(open(\"../input/otto-full-optimized-memory-footprint/id2type.pkl\",'rb'))\ntype2id = pickle.load(open(\"../input/otto-full-optimized-memory-footprint/type2id.pkl\",'rb'))\n\ntrain = pl.scan_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')#, chunksize=1000)\ntest = pl.scan_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n# merged_pl = pl.concat([train,test])\n\npairs = pl.read_parquet('../input/ottorecommendationsimplepairs/pairs.pq')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:41.647598Z","iopub.execute_input":"2022-12-29T11:42:41.648046Z","iopub.status.idle":"2022-12-29T11:42:51.910959Z","shell.execute_reply.started":"2022-12-29T11:42:41.648000Z","shell.execute_reply":"2022-12-29T11:42:51.909819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pairs to integers","metadata":{}},{"cell_type":"code","source":"cardinality_of_aid = pairs.select('aid').max()[0,0]","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:51.915382Z","iopub.execute_input":"2022-12-29T11:42:51.916346Z","iopub.status.idle":"2022-12-29T11:42:51.996488Z","shell.execute_reply.started":"2022-12-29T11:42:51.916277Z","shell.execute_reply":"2022-12-29T11:42:51.995452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtrain = pairs.select('aid').to_numpy()\nYtrain = pairs.select('aid1').to_numpy()\n# Xtrain = [pair2id[tuple(z)] for z in Xtrain]\n# Ytrain = [pair2id[tuple(z)] for z in Ytrain]\n# Xtrain = np.array(Xtrain)\n# Ytrain = np.array(Ytrain)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:51.998176Z","iopub.execute_input":"2022-12-29T11:42:51.998571Z","iopub.status.idle":"2022-12-29T11:42:53.711609Z","shell.execute_reply.started":"2022-12-29T11:42:51.998531Z","shell.execute_reply":"2022-12-29T11:42:53.710506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# using keras for embeddings","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Embedding,Input,BatchNormalization\nfrom tensorflow import keras\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow import GradientTape\nfrom tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.random import shuffle\nimport tensorflow as tf\n\nimport time","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:53.716480Z","iopub.execute_input":"2022-12-29T11:42:53.718971Z","iopub.status.idle":"2022-12-29T11:42:59.800389Z","shell.execute_reply.started":"2022-12-29T11:42:53.718932Z","shell.execute_reply":"2022-12-29T11:42:59.799205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Batchsize is quite large, but with smaller batches we don't get through the entire dataset for more than a day.","metadata":{}},{"cell_type":"code","source":"num_tokens = cardinality_of_aid+1\ndim_embedding = 32\nbatchsize=1024\nembedding_layer = Embedding(num_tokens,dim_embedding)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:59.801741Z","iopub.execute_input":"2022-12-29T11:42:59.802702Z","iopub.status.idle":"2022-12-29T11:42:59.839980Z","shell.execute_reply.started":"2022-12-29T11:42:59.802615Z","shell.execute_reply":"2022-12-29T11:42:59.838818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = tf.data.Dataset.from_tensor_slices((Xtrain,Ytrain))\ndataset = dataset.shuffle(10*batchsize)\ndataset = dataset.batch(batchsize)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:42:59.841394Z","iopub.execute_input":"2022-12-29T11:42:59.842003Z","iopub.status.idle":"2022-12-29T11:43:03.741125Z","shell.execute_reply.started":"2022-12-29T11:42:59.841968Z","shell.execute_reply":"2022-12-29T11:43:03.739725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Input(shape=(1,),batch_size=batchsize))\nmodel.add(embedding_layer)\n# model.add(BatchNormalization())\n\nmodel.compile()\n\nbce = BinaryCrossentropy(from_logits=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:43:03.743761Z","iopub.execute_input":"2022-12-29T11:43:03.744223Z","iopub.status.idle":"2022-12-29T11:43:03.820106Z","shell.execute_reply.started":"2022-12-29T11:43:03.744186Z","shell.execute_reply":"2022-12-29T11:43:03.819083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = Adam(learning_rate=(0.001 /64*batchsize/64))\n# dataset = dataset_gen(dummy_x, dummy_y, batchsize=batchsize)\n# Iterate over the batches of a dataset.\ni=0\ntotal = len(Xtrain)\ntstart = time.time()\nprint(\"Start Training\")\n# dot = Dot(2)\nfor x, y in dataset:\n    # Open a GradientTape.\n    \n    with GradientTape() as tape:\n#         print(\"model eval\")\n        # Forward pass.\n        xembed = model(x)\n        # positive match\n        yembed = model(y)\n        positive_overlap = tf.matmul(xembed,yembed, transpose_b=True)\n        \n        # negative match\n        yembed_negative = model(tf.random.shuffle(y))# tf.transpose(tf.random.shuffle(tf.transpose(yembed),))\n        negative_overlap = tf.matmul(xembed,yembed_negative, transpose_b=True)\n        \n        y_true = tf.concat([tf.ones_like(positive_overlap),tf.zeros_like(negative_overlap)],axis=0)\n        y_pred = tf.concat([positive_overlap,negative_overlap],axis=0)\n        \n        # Loss value for this batch.\n#         print(\"loss eval\")\n        loss_value = bce(y_true,y_pred)\n        \n\n    # Get gradients of loss wrt the weights.\n#     print(\"grad\")\n    \n    gradients = tape.gradient(loss_value, model.trainable_weights)\n#     print(\"apply\")\n    # Update the weights of the model.\n    optimizer.apply_gradients(zip(gradients, model.trainable_weights))\n    \n    if i%100==0:\n        print(f\"Batch {i*batchsize}/{total} in {time.time()-tstart}s remaining: {np.round((total-(i*batchsize))/np.max([(i*batchsize/(time.time()-tstart)),1]),1)}\", end=\"\\t\")\n        print(f\"loss: {np.round(np.mean(loss_value.numpy()),4)}\", end=\"\\n\")\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-12-29T11:43:24.892261Z","iopub.execute_input":"2022-12-29T11:43:24.892878Z","iopub.status.idle":"2022-12-29T12:22:50.276634Z","shell.execute_reply.started":"2022-12-29T11:43:24.892837Z","shell.execute_reply":"2022-12-29T12:22:50.275158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"/kaggle/working/matrix_factorization_model\")","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:22:53.473751Z","iopub.execute_input":"2022-12-29T12:22:53.474110Z","iopub.status.idle":"2022-12-29T12:22:54.494925Z","shell.execute_reply.started":"2022-12-29T12:22:53.474080Z","shell.execute_reply":"2022-12-29T12:22:54.493863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"def tuple_list_to_submit(x):\n    sess = int(x['session'])\n    res = pd.DataFrame(index=[f\"{sess}_clicks\",f\"{sess}_carts\",f\"{sess}_orders\"])\n    res['labels']=''\n    counts = {'clicks':0,'carts':0,'orders':0}\n    if x['new_aids'] is not None:\n        for aid,typ in x['new_aids']:\n            if counts[id2type[typ]]<20:\n                res.loc[f\"{sess}_{id2type[typ]}\",'labels'] += f\"{int(aid)} \"\n                counts[id2type[typ]]+=1\n    for idx in [f\"{sess}_clicks\",f\"{sess}_carts\",f\"{sess}_orders\"]:\n        res.loc[idx,'labels'].strip()\n    return res","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:04:03.323192Z","iopub.execute_input":"2022-12-18T01:04:03.324591Z","iopub.status.idle":"2022-12-18T01:04:03.340226Z","shell.execute_reply.started":"2022-12-18T01:04:03.324539Z","shell.execute_reply":"2022-12-18T01:04:03.338623Z"}}},{"cell_type":"raw","source":"result = pd.concat(last_events.reset_index().apply(tuple_list_to_submit,axis=1).values, axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:05:48.331280Z","iopub.execute_input":"2022-12-18T01:05:48.332053Z","iopub.status.idle":"2022-12-18T01:05:48.374099Z","shell.execute_reply.started":"2022-12-18T01:05:48.332014Z","shell.execute_reply":"2022-12-18T01:05:48.372612Z"}}},{"cell_type":"raw","source":"result = result.reset_index()\nresult.columns = ['session_type','labels']\nresult.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T01:08:21.131345Z","iopub.execute_input":"2022-12-18T01:08:21.131803Z","iopub.status.idle":"2022-12-18T01:08:21.145143Z","shell.execute_reply.started":"2022-12-18T01:08:21.131769Z","shell.execute_reply":"2022-12-18T01:08:21.144214Z"}}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}