{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-13T17:52:05.425735Z","iopub.execute_input":"2023-01-13T17:52:05.426279Z","iopub.status.idle":"2023-01-13T17:52:05.457414Z","shell.execute_reply.started":"2023-01-13T17:52:05.426159Z","shell.execute_reply":"2023-01-13T17:52:05.456320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport json","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:52:05.459047Z","iopub.execute_input":"2023-01-13T17:52:05.459437Z","iopub.status.idle":"2023-01-13T17:52:05.465238Z","shell.execute_reply.started":"2023-01-13T17:52:05.459402Z","shell.execute_reply":"2023-01-13T17:52:05.463974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load 1 million record from the data \n# you can change this number based on your needs","metadata":{}},{"cell_type":"code","source":"f=open('/kaggle/input/otto-recommender-system/train.jsonl', 'r').readlines()[:1000]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:52:05.467029Z","iopub.execute_input":"2023-01-13T17:52:05.467458Z","iopub.status.idle":"2023-01-13T17:54:40.710926Z","shell.execute_reply.started":"2023-01-13T17:52:05.467421Z","shell.execute_reply":"2023-01-13T17:54:40.709676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# make a loop on the data to arrnge it \n# add every single object in a list \n# all you will need in a recommendation system model these features and you can change to what you need","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\nrecords=[]\nfor line in tqdm(f):\n        # Parse the JSON object\n    obj = json.loads(line)\n    session=obj['session'] \n    for event in obj['events']:\n        # Get the aid for the event\n        aid = event['aid']\n        type=event['type']\n        record=[session,aid,type]\n        records.append(record)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:40.713819Z","iopub.execute_input":"2023-01-13T17:54:40.714517Z","iopub.status.idle":"2023-01-13T17:54:40.990465Z","shell.execute_reply.started":"2023-01-13T17:54:40.714478Z","shell.execute_reply":"2023-01-13T17:54:40.989209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# convert the list into dataframe","metadata":{}},{"cell_type":"code","source":"df=pd.DataFrame(records,columns=['session_id','aid','type'])","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:40.992116Z","iopub.execute_input":"2023-01-13T17:54:40.993300Z","iopub.status.idle":"2023-01-13T17:54:41.053634Z","shell.execute_reply.started":"2023-01-13T17:54:40.993250Z","shell.execute_reply":"2023-01-13T17:54:41.052311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:41.055066Z","iopub.execute_input":"2023-01-13T17:54:41.055469Z","iopub.status.idle":"2023-01-13T17:54:41.086642Z","shell.execute_reply.started":"2023-01-13T17:54:41.055433Z","shell.execute_reply":"2023-01-13T17:54:41.085451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:41.087995Z","iopub.execute_input":"2023-01-13T17:54:41.088373Z","iopub.status.idle":"2023-01-13T17:54:41.121334Z","shell.execute_reply.started":"2023-01-13T17:54:41.088340Z","shell.execute_reply":"2023-01-13T17:54:41.120118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:41.122831Z","iopub.execute_input":"2023-01-13T17:54:41.123200Z","iopub.status.idle":"2023-01-13T17:54:41.137746Z","shell.execute_reply.started":"2023-01-13T17:54:41.123122Z","shell.execute_reply":"2023-01-13T17:54:41.136037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:41.139748Z","iopub.execute_input":"2023-01-13T17:54:41.140478Z","iopub.status.idle":"2023-01-13T17:54:41.158478Z","shell.execute_reply.started":"2023-01-13T17:54:41.140426Z","shell.execute_reply":"2023-01-13T17:54:41.157285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn import preprocessing\n  \nlabel_encoder = preprocessing.LabelEncoder()\n  \ndf['type']= label_encoder.fit_transform(df['type'])\n  \ndf['type'].unique()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T17:54:41.161320Z","iopub.execute_input":"2023-01-13T17:54:41.161697Z","iopub.status.idle":"2023-01-13T17:54:42.235821Z","shell.execute_reply.started":"2023-01-13T17:54:41.161663Z","shell.execute_reply":"2023-01-13T17:54:42.233245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matrix = df.pivot_table(values='type', index='session_id', columns='aid', fill_value=4)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:01:14.978429Z","iopub.execute_input":"2023-01-13T18:01:14.979946Z","iopub.status.idle":"2023-01-13T18:01:24.355453Z","shell.execute_reply.started":"2023-01-13T18:01:14.979887Z","shell.execute_reply":"2023-01-13T18:01:24.354221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matrix.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:01:36.824320Z","iopub.execute_input":"2023-01-13T18:01:36.824752Z","iopub.status.idle":"2023-01-13T18:01:36.852027Z","shell.execute_reply.started":"2023-01-13T18:01:36.824718Z","shell.execute_reply":"2023-01-13T18:01:36.850647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matrix.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:01:43.972249Z","iopub.execute_input":"2023-01-13T18:01:43.976082Z","iopub.status.idle":"2023-01-13T18:01:44.001920Z","shell.execute_reply.started":"2023-01-13T18:01:43.975981Z","shell.execute_reply":"2023-01-13T18:01:43.999562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=matrix.T","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:01:46.507253Z","iopub.execute_input":"2023-01-13T18:01:46.510703Z","iopub.status.idle":"2023-01-13T18:01:46.694132Z","shell.execute_reply.started":"2023-01-13T18:01:46.510348Z","shell.execute_reply":"2023-01-13T18:01:46.690833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import TruncatedSVD\nSVD = TruncatedSVD(n_components=10)\ndecomposed_matrix = SVD.fit_transform(X)\ndecomposed_matrix.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:01:50.669014Z","iopub.execute_input":"2023-01-13T18:01:50.673164Z","iopub.status.idle":"2023-01-13T18:01:59.398923Z","shell.execute_reply.started":"2023-01-13T18:01:50.672904Z","shell.execute_reply":"2023-01-13T18:01:59.391345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = np.corrcoef(decomposed_matrix)\ncorrelation_matrix.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:04:44.863941Z","iopub.execute_input":"2023-01-13T18:04:44.864963Z","iopub.status.idle":"2023-01-13T18:05:18.987507Z","shell.execute_reply.started":"2023-01-13T18:04:44.864916Z","shell.execute_reply":"2023-01-13T18:05:18.986165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.index[100]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:24:22.476099Z","iopub.execute_input":"2023-01-13T18:24:22.476580Z","iopub.status.idle":"2023-01-13T18:24:22.485498Z","shell.execute_reply.started":"2023-01-13T18:24:22.476544Z","shell.execute_reply":"2023-01-13T18:24:22.484068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 4284\n\naids = list(X.index)\nproduct_ID = aids.index(i)\nproduct_ID","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:26:56.488275Z","iopub.execute_input":"2023-01-13T18:26:56.488782Z","iopub.status.idle":"2023-01-13T18:26:56.503752Z","shell.execute_reply.started":"2023-01-13T18:26:56.488744Z","shell.execute_reply":"2023-01-13T18:26:56.502256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_product_ID = correlation_matrix[product_ID]\ncorrelation_product_ID.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:26:59.708081Z","iopub.execute_input":"2023-01-13T18:26:59.708582Z","iopub.status.idle":"2023-01-13T18:26:59.716609Z","shell.execute_reply.started":"2023-01-13T18:26:59.708540Z","shell.execute_reply":"2023-01-13T18:26:59.715405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Recommend = list(X.index[correlation_product_ID >= 0.99])\n\n# Removes the item already bought by the customer\nRecommend.remove(i) \n\nRecommend[0:9]","metadata":{"execution":{"iopub.status.busy":"2023-01-13T18:27:50.785948Z","iopub.execute_input":"2023-01-13T18:27:50.786456Z","iopub.status.idle":"2023-01-13T18:27:50.803130Z","shell.execute_reply.started":"2023-01-13T18:27:50.786418Z","shell.execute_reply":"2023-01-13T18:27:50.801980Z"},"trusted":true},"execution_count":null,"outputs":[]}]}