{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Based on\n# https://www.kaggle.com/code/dpalbrecht/fast-co-visitation-matrix\n\n# https://www.kaggle.com/code/vslaykovsky/co-visitation-matrix.\n# https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:32.902385Z","iopub.execute_input":"2023-01-09T04:43:32.902835Z","iopub.status.idle":"2023-01-09T04:43:32.909677Z","shell.execute_reply.started":"2023-01-09T04:43:32.902800Z","shell.execute_reply":"2023-01-09T04:43:32.907680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/otto-full-optimized-memory-footprint/","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:09.189396Z","iopub.execute_input":"2023-01-09T04:43:09.190368Z","iopub.status.idle":"2023-01-09T04:43:10.351489Z","shell.execute_reply.started":"2023-01-09T04:43:09.190328Z","shell.execute_reply":"2023-01-09T04:43:10.349802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom collections import defaultdict, Counter\nfrom tqdm import tqdm #progress bar\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-09T04:43:36.275839Z","iopub.execute_input":"2023-01-09T04:43:36.276380Z","iopub.status.idle":"2023-01-09T04:43:49.457948Z","shell.execute_reply.started":"2023-01-09T04:43:36.276332Z","shell.execute_reply":"2023-01-09T04:43:49.456617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train = train.groupby('session').tail(30).reset_index(drop = True)\n#test = test.groupby('session').tail(30).reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:29.158034Z","iopub.status.idle":"2023-01-09T04:43:29.159017Z","shell.execute_reply.started":"2023-01-09T04:43:29.158692Z","shell.execute_reply":"2023-01-09T04:43:29.158727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(columns='type',inplace=True)\ntest.drop(columns='type',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:49.460492Z","iopub.execute_input":"2023-01-09T04:43:49.460911Z","iopub.status.idle":"2023-01-09T04:43:51.507725Z","shell.execute_reply.started":"2023-01-09T04:43:49.460865Z","shell.execute_reply":"2023-01-09T04:43:51.506185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.session = train.session.astype(np.int32)\ntrain.aid = train.aid.astype(np.int32)\ntrain.ts /= 1000\ntrain.ts = train.ts.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:51.509764Z","iopub.execute_input":"2023-01-09T04:43:51.510324Z","iopub.status.idle":"2023-01-09T04:43:59.658896Z","shell.execute_reply.started":"2023-01-09T04:43:51.510252Z","shell.execute_reply":"2023-01-09T04:43:59.657712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2023-01-09T05:08:50.745505Z","iopub.execute_input":"2023-01-09T05:08:50.746183Z","iopub.status.idle":"2023-01-09T05:08:50.770985Z","shell.execute_reply.started":"2023-01-09T05:08:50.746130Z","shell.execute_reply":"2023-01-09T05:08:50.769320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.aid.unique().size)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:59.661021Z","iopub.execute_input":"2023-01-09T04:43:59.661547Z","iopub.status.idle":"2023-01-09T04:44:04.160512Z","shell.execute_reply.started":"2023-01-09T04:43:59.661508Z","shell.execute_reply":"2023-01-09T04:44:04.158908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note: add rng factor that picks only ~2_000_000 rows of train.","metadata":{}},{"cell_type":"code","source":"next_aids = defaultdict(Counter)\n\ndef upd_aid_pair(df, chunk_size = 30_000, day_length = 24*60*60):\n    #only take the last 30 entries per session\n    df= df.groupby('session').tail(30).reset_index(drop = True)\n        \n    for i in tqdm(range(0, df.shape[0], chunk_size)):\n        #get current chunk\n        chunk = df.loc[i:min(df.shape[0]-1, i+chunk_size-1)].reset_index(drop = True)\n        \n        #chunk.drop(columns = 'type', inplace = True)\n        \n        #merge with itself to get aid pairs\n        chunk = chunk.merge(chunk, on='session')\n        \n        #remove entries that are themselfs\n        chunk = chunk[chunk.aid_x != chunk.aid_y]\n        \n        #get days between entries\n        chunk['days_passed'] = (chunk.ts_x - chunk.ts_y) / day_length\n        \n        #only keep entries that are with 1 day of each other\n        chunk = chunk[(0 <= chunk.days_passed) & (chunk.days_passed <= 1)]\n        \n        #remove duplicates\n        chunk.drop_duplicates(subset=['session','aid_x','aid_y'], inplace = True)\n        for aid_x, aid_y in zip(chunk.aid_x, chunk.aid_y):\n            next_aids[aid_x][aid_y] += 1\n        \nupd_aid_pair(train)\nupd_aid_pair(test)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:51:22.289728Z","iopub.execute_input":"2023-01-09T04:51:22.291049Z","iopub.status.idle":"2023-01-09T05:05:58.883186Z","shell.execute_reply.started":"2023-01-09T04:51:22.290984Z","shell.execute_reply":"2023-01-09T05:05:58.879324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(next_aids)","metadata":{"execution":{"iopub.status.busy":"2023-01-09T05:05:58.884973Z","iopub.status.idle":"2023-01-09T05:05:58.885613Z","shell.execute_reply.started":"2023-01-09T05:05:58.885314Z","shell.execute_reply":"2023-01-09T05:05:58.885341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aids_to_int = defaultdict()\nint_to_aids = defaultdict()\ntotal_number_counter = 0\nfor i in tqdm(next_aids):\n    aids_to_int[i] = total_number_counter\n    int_to_aids[total_number_counter] = i\n    total_number_counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:29.176918Z","iopub.status.idle":"2023-01-09T04:43:29.177809Z","shell.execute_reply.started":"2023-01-09T04:43:29.177501Z","shell.execute_reply":"2023-01-09T04:43:29.177531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#total_number_counter = 1_000_000\n#aid_item = np.random.uniform(low=0.1,high=0.9, size=(3,total_number_counter))\n#aid_item = pd.DataFrame(np.random.uniform(low=0.1,high=0.9, size=(total_number_counter,3)))\n#product = aid_item.dot(aid_item.T)\n#product","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:29.179435Z","iopub.status.idle":"2023-01-09T04:43:29.180294Z","shell.execute_reply.started":"2023-01-09T04:43:29.179967Z","shell.execute_reply":"2023-01-09T04:43:29.179997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub    ","metadata":{"execution":{"iopub.status.busy":"2023-01-09T04:43:29.181920Z","iopub.status.idle":"2023-01-09T04:43:29.182822Z","shell.execute_reply.started":"2023-01-09T04:43:29.182509Z","shell.execute_reply":"2023-01-09T04:43:29.182542Z"},"trusted":true},"execution_count":null,"outputs":[]}]}