{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-28T04:21:57.120058Z","iopub.execute_input":"2022-11-28T04:21:57.120550Z","iopub.status.idle":"2022-11-28T04:21:57.178045Z","shell.execute_reply.started":"2022-11-28T04:21:57.120464Z","shell.execute_reply":"2022-11-28T04:21:57.177098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspark --user","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:21:57.181563Z","iopub.execute_input":"2022-11-28T04:21:57.182522Z","iopub.status.idle":"2022-11-28T04:22:43.771785Z","shell.execute_reply.started":"2022-11-28T04:21:57.182477Z","shell.execute_reply":"2022-11-28T04:22:43.770576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nfrom pyspark.sql.functions import count, desc , col, max\nfrom pyspark.ml.feature import  StringIndexer\nfrom pyspark.ml import Pipeline\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.ml.tuning import TrainValidationSplit, ParamGridBuilder","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:22:43.773572Z","iopub.execute_input":"2022-11-28T04:22:43.774318Z","iopub.status.idle":"2022-11-28T04:22:43.917299Z","shell.execute_reply.started":"2022-11-28T04:22:43.774266Z","shell.execute_reply":"2022-11-28T04:22:43.916318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spark=SparkSession.builder.appName('LastFm').getOrCreate()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:22:43.919671Z","iopub.execute_input":"2022-11-28T04:22:43.919951Z","iopub.status.idle":"2022-11-28T04:22:51.028245Z","shell.execute_reply.started":"2022-11-28T04:22:43.919926Z","shell.execute_reply":"2022-11-28T04:22:51.027111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc = spark.sparkContext\npath = '/kaggle/input/otto-recommender-system/train.jsonl'\ndf = spark.read.json(path)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:22:51.033654Z","iopub.execute_input":"2022-11-28T04:22:51.035882Z","iopub.status.idle":"2022-11-28T04:25:34.398005Z","shell.execute_reply.started":"2022-11-28T04:22:51.035836Z","shell.execute_reply":"2022-11-28T04:25:34.397014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#uncomment if importing via zipfile\n'''\n#import zipfile module\nfrom zipfile import ZipFile\n\nwith ZipFile('otto-recommender-system.zip', 'r') as f:\n\n#extract in current directory\n    f.extractall('/content') # saves data in temporary folder content\n'''","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:34.399246Z","iopub.execute_input":"2022-11-28T04:25:34.399563Z","iopub.status.idle":"2022-11-28T04:25:34.410132Z","shell.execute_reply.started":"2022-11-28T04:25:34.399534Z","shell.execute_reply":"2022-11-28T04:25:34.409077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nfrom pyspark.sql.functions import count, desc , col, max\nfrom pyspark.ml.feature import  StringIndexer\nfrom pyspark.ml import Pipeline\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.ml.tuning import TrainValidationSplit, ParamGridBuilder\nfrom pyspark.sql.functions import explode","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:34.411769Z","iopub.execute_input":"2022-11-28T04:25:34.412664Z","iopub.status.idle":"2022-11-28T04:25:34.420275Z","shell.execute_reply.started":"2022-11-28T04:25:34.412624Z","shell.execute_reply":"2022-11-28T04:25:34.419610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_file_path=\"/kaggle/input/otto-recommender-system/train.jsonl\"\ntrain_data=spark.read.format('json').option('header',True).option('inferschema',True).load(train_file_path)\n\ntest_file_path=\"/kaggle/input/otto-recommender-system/test.jsonl\"\ntest_data=spark.read.format('json').option('header',True).option('inferschema',True).load(test_file_path)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:34.422005Z","iopub.execute_input":"2022-11-28T04:25:34.422781Z","iopub.status.idle":"2022-11-28T04:25:34.433823Z","shell.execute_reply.started":"2022-11-28T04:25:34.422744Z","shell.execute_reply":"2022-11-28T04:25:34.432951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:34.435231Z","iopub.execute_input":"2022-11-28T04:25:34.436018Z","iopub.status.idle":"2022-11-28T04:25:34.498037Z","shell.execute_reply.started":"2022-11-28T04:25:34.435976Z","shell.execute_reply":"2022-11-28T04:25:34.496779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df.withColumn(\"events\",explode(\"events\"))\ndf1.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:34.502047Z","iopub.execute_input":"2022-11-28T04:25:34.502759Z","iopub.status.idle":"2022-11-28T04:25:35.472669Z","shell.execute_reply.started":"2022-11-28T04:25:34.502721Z","shell.execute_reply":"2022-11-28T04:25:35.471254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_df = df1.select('session')\nsession_df.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:35.473652Z","iopub.execute_input":"2022-11-28T04:25:35.473977Z","iopub.status.idle":"2022-11-28T04:25:35.749944Z","shell.execute_reply.started":"2022-11-28T04:25:35.473945Z","shell.execute_reply":"2022-11-28T04:25:35.748813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_df = df1.select(\"events.*\")\nevents_df.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:35.751136Z","iopub.execute_input":"2022-11-28T04:25:35.753075Z","iopub.status.idle":"2022-11-28T04:25:36.011826Z","shell.execute_reply.started":"2022-11-28T04:25:35.753029Z","shell.execute_reply":"2022-11-28T04:25:36.010739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = events_df.join(session_df)\ndf2.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:25:36.012922Z","iopub.execute_input":"2022-11-28T04:25:36.013281Z","iopub.status.idle":"2022-11-28T04:25:36.703837Z","shell.execute_reply.started":"2022-11-28T04:25:36.013245Z","shell.execute_reply":"2022-11-28T04:25:36.702871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spark.conf.set(\"spark.sql.execution.arrow.pyspark.fallback.enabled\", \"true\")\n\ndfpd = df2.limit(8000000).toPandas()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:36:44.036483Z","iopub.execute_input":"2022-11-28T04:36:44.037181Z","iopub.status.idle":"2022-11-28T04:38:28.283177Z","shell.execute_reply.started":"2022-11-28T04:36:44.037144Z","shell.execute_reply":"2022-11-28T04:38:28.282119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfpd.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:35:16.499301Z","iopub.execute_input":"2022-11-28T04:35:16.500005Z","iopub.status.idle":"2022-11-28T04:35:16.518630Z","shell.execute_reply.started":"2022-11-28T04:35:16.499969Z","shell.execute_reply":"2022-11-28T04:35:16.517496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfpd.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:35:19.589621Z","iopub.execute_input":"2022-11-28T04:35:19.589980Z","iopub.status.idle":"2022-11-28T04:35:19.597029Z","shell.execute_reply.started":"2022-11-28T04:35:19.589949Z","shell.execute_reply":"2022-11-28T04:35:19.595984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfpd.to_csv('train_df.csv', index = True)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T04:35:19.940520Z","iopub.execute_input":"2022-11-28T04:35:19.940872Z","iopub.status.idle":"2022-11-28T04:35:35.852650Z","shell.execute_reply.started":"2022-11-28T04:35:19.940845Z","shell.execute_reply":"2022-11-28T04:35:35.851675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}