{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport wandb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-02T04:31:48.330363Z","iopub.execute_input":"2022-12-02T04:31:48.330756Z","iopub.status.idle":"2022-12-02T04:31:48.341311Z","shell.execute_reply.started":"2022-12-02T04:31:48.330727Z","shell.execute_reply":"2022-12-02T04:31:48.339668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip3 install cudf \n# !pip3 install cupy","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:31:48.347828Z","iopub.execute_input":"2022-12-02T04:31:48.348343Z","iopub.status.idle":"2022-12-02T04:31:48.353190Z","shell.execute_reply.started":"2022-12-02T04:31:48.348312Z","shell.execute_reply":"2022-12-02T04:31:48.351673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# General Libraries","metadata":{}},{"cell_type":"code","source":"\nimport os\nimport re\nimport gc\nimport wandb\nimport random\nimport math\nfrom tqdm import tqdm\nfrom pprint import pprint\nfrom time import time\nfrom datetime import datetime\nimport itertools\nimport warnings\nimport pandas as pd\nimport numpy as np\n\n# For the Visuals\nimport seaborn as sns\nimport matplotlib as mpl\nfrom matplotlib import cm\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.patches import Rectangle\nfrom IPython.display import display_html\nplt.rcParams.update({'font.size': 16})\n\n# RAPIDS\n# import cudf\n# import cupy\n\n# Environment check\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"WANDB_SILENT\"] = \"true\"\nCONFIG = {'competition': 'Otto', '_wandb_kernel': 'aot'} \n# Custom colors\nclass clr:\n    S = '\\033[1m' + '\\033[91m'\n    E = '\\033[0m'\n    \nmy_colors = [\"#f3afc2\", \"#a86c4a\", \"#7f5c10\", \"#d79a7b\", \n             \"#ab883e\", \"#7a7300\", \"#004c00\"]\nCMAP1 = ListedColormap(my_colors)\n\nprint(clr.S+\"Notebook Color Schemes:\"+clr.E)\nsns.palplot(sns.color_palette(my_colors))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:31:48.363850Z","iopub.execute_input":"2022-12-02T04:31:48.364750Z","iopub.status.idle":"2022-12-02T04:31:48.679318Z","shell.execute_reply.started":"2022-12-02T04:31:48.364714Z","shell.execute_reply":"2022-12-02T04:31:48.678372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Secrets\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"WandbKy\")\n\n! wandb login $secret_value_0","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:31:48.683134Z","iopub.execute_input":"2022-12-02T04:31:48.683441Z","iopub.status.idle":"2022-12-02T04:31:51.580573Z","shell.execute_reply.started":"2022-12-02T04:31:48.683414Z","shell.execute_reply":"2022-12-02T04:31:51.579353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# run = wandb.init(project='MultiObjective Recommender System', name='CoverPhoto', config=CONFIG)\n# cover = plt.imread(\"../input/otto-helper-data/recsys_cover.png\")\n# wandb.log({\"cover\": wandb.Image(cover)})\n# wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:31:51.582498Z","iopub.execute_input":"2022-12-02T04:31:51.582780Z","iopub.status.idle":"2022-12-02T04:31:51.587707Z","shell.execute_reply.started":"2022-12-02T04:31:51.582753Z","shell.execute_reply":"2022-12-02T04:31:51.586580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Experiment\n# run = wandb.init(project='Otto', \n#                  name='base_info', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:33:00.301362Z","iopub.execute_input":"2022-12-02T04:33:00.301767Z","iopub.status.idle":"2022-12-02T04:33:00.306055Z","shell.execute_reply.started":"2022-12-02T04:33:00.301728Z","shell.execute_reply":"2022-12-02T04:33:00.305055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the JSON File\n#Loading the training datasets \n# READING the whole dataset leads to out of memory and restart the notebooks\n# TrainMORS=pd.read_json(\"/kaggle/input/otto-recommender-system/train.jsonl\", lines=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:31:53.339231Z","iopub.status.idle":"2022-12-02T04:31:53.339570Z","shell.execute_reply.started":"2022-12-02T04:31:53.339397Z","shell.execute_reply":"2022-12-02T04:31:53.339413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"debug = True\nif debug:\n    df_train = pd.read_json(r'/kaggle/input/otto-recommender-system/train.jsonl',\n                            nrows=1000 ,  lines=True,orient='records')\n    df_test  = pd.read_json(r'/kaggle/input/otto-recommender-system/test.jsonl',\n                            nrows=300 ,  lines=True,orient='records')\nelse:\n    df_train = pd.read_json(r'/kaggle/input/otto-recommender-system/train.jsonl',nrows=1000 ,  lines=True)\n    df_test  = pd.read_json(r'/kaggle/input/otto-recommender-system/test.jsonl',nrows=300 ,  lines=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:33:04.635088Z","iopub.execute_input":"2022-12-02T04:33:04.635473Z","iopub.status.idle":"2022-12-02T04:33:04.783539Z","shell.execute_reply.started":"2022-12-02T04:33:04.635442Z","shell.execute_reply":"2022-12-02T04:33:04.782594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### OR","metadata":{}},{"cell_type":"code","source":"chunksize = 20_000\n\ntrain_data = pd.read_json(\"../input/otto-recommender-system/train.jsonl\", lines=True, chunksize=chunksize)\ntest_data = pd.read_json(\"../input/otto-recommender-system/test.jsonl\", lines=True, chunksize=chunksize)\nsample_submission = pd.read_csv(\"../input/otto-recommender-system/sample_submission.csv\", chunksize=chunksize)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:34:01.303529Z","iopub.execute_input":"2022-12-02T04:34:01.303897Z","iopub.status.idle":"2022-12-02T04:34:01.314598Z","shell.execute_reply.started":"2022-12-02T04:34:01.303870Z","shell.execute_reply":"2022-12-02T04:34:01.313354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's examine number of sessions in train and test","metadata":{}},{"cell_type":"code","source":"with open('../input/otto-recommender-system/train.jsonl', 'r') as f:\n    print(f\"Train: {len(f.readlines()):,} lines\")\nwith open('../input/otto-recommender-system/test.jsonl', 'r') as f:\n    print(f\"Test {len(f.readlines()):,} lines\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:34:08.266682Z","iopub.execute_input":"2022-12-02T04:34:08.270077Z","iopub.status.idle":"2022-12-02T04:35:38.763275Z","shell.execute_reply.started":"2022-12-02T04:34:08.269836Z","shell.execute_reply":"2022-12-02T04:35:38.761900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As it expected, dataset is realy big. So let's load the first chunk of data to examine structure of the datset","metadata":{}},{"cell_type":"code","source":"train_data_chunk = train_data.__next__()\ntrain_data_chunk","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:35:38.765770Z","iopub.execute_input":"2022-12-02T04:35:38.766831Z","iopub.status.idle":"2022-12-02T04:35:40.935431Z","shell.execute_reply.started":"2022-12-02T04:35:38.766791Z","shell.execute_reply":"2022-12-02T04:35:40.934520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Extract data to work with","metadata":{}},{"cell_type":"code","source":"indices = [i for i in range(100)]\nrandom.shuffle(indices)\nindices = indices[:3]\nprint(f\"Chunks chosen: {indices}\")\n\nchunks_of_train = []\nfor idx, chunk in enumerate(train_data):\n    if idx in indices:\n        chunks_of_train.append(chunk)\n    if idx > max(indices):\n        break\n\nchunks_of_train = pd.concat(chunks_of_train)\nprint(chunks_of_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:35:40.936482Z","iopub.execute_input":"2022-12-02T04:35:40.937100Z","iopub.status.idle":"2022-12-02T04:37:27.092113Z","shell.execute_reply.started":"2022-12-02T04:35:40.937072Z","shell.execute_reply":"2022-12-02T04:37:27.090777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_dict = {\n    \"session\": [],\n    \"aid\": [],\n    \"ts\": [],\n    \"type\": [],\n}\n\nfor _, row in chunks_of_train.iterrows():\n    for event in row[\"events\"]:\n        events_dict[\"session\"].append(row[\"session\"])\n        events_dict[\"aid\"].append(event[\"aid\"])\n        events_dict[\"ts\"].append(event[\"ts\"])\n        events_dict[\"type\"].append(event[\"type\"])\n\ntrain_part = pd.DataFrame(events_dict)\ntrain_part","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:37:27.094831Z","iopub.execute_input":"2022-12-02T04:37:27.095395Z","iopub.status.idle":"2022-12-02T04:37:40.336741Z","shell.execute_reply.started":"2022-12-02T04:37:27.095365Z","shell.execute_reply":"2022-12-02T04:37:40.335790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_sessions = train_part[\"session\"].nunique()\nn_events = train_part.shape[0]\n\nprint(f\"Number of sessions: {n_sessions}\")\nprint(f\"Number of events: {n_events}\")\nprint(f\"Mean number of events in session: {n_events/n_sessions}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T04:37:40.337926Z","iopub.execute_input":"2022-12-02T04:37:40.338423Z","iopub.status.idle":"2022-12-02T04:37:40.354188Z","shell.execute_reply.started":"2022-12-02T04:37:40.338397Z","shell.execute_reply":"2022-12-02T04:37:40.353269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Explotary Data Analysis**\nThe Notebook Taken From [mohdmuttalib](https://www.kaggle.com/code/mohdmuttalib/otto-eda) \n\nThanks [mohdmuttalib](https://www.kaggle.com/code/mohdmuttalib/otto-eda) \n\n**I will complete the notebook soon!**","metadata":{}}]}