{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## cuDF VS Pandas\n\nAll credit of the data manipulation goes to @GIBA and his notebook [Article_id pairs in 3s using cuDF](https://www.kaggle.com/code/titericz/article-id-pairs-in-3s-using-cudf).\n\n\nIn this notebook we just try to compare the performance of cuDF compared to pandas. It is not exhaustive but it can give an idea whether it is worth to use cuDF (GPU) for data manipulation instead of pandas (CPU). Since the API that they provide are quite identical they are very easy to compare.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cudf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-15T17:00:04.936788Z","iopub.execute_input":"2022-04-15T17:00:04.937245Z","iopub.status.idle":"2022-04-15T17:00:04.941056Z","shell.execute_reply.started":"2022-04-15T17:00:04.937210Z","shell.execute_reply":"2022-04-15T17:00:04.940265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport gc\n\ndef mytimeit(f, n_exec=5, **kwargs):\n    times = []\n    for i in range(0, n_exec):\n        t1 = time.time()\n        res = f(**kwargs)\n        t2 = time.time()\n        times.append(t2 - t1)\n        gc.collect()\n    return times, res","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:00:05.247953Z","iopub.execute_input":"2022-04-15T17:00:05.248804Z","iopub.status.idle":"2022-04-15T17:00:05.254740Z","shell.execute_reply.started":"2022-04-15T17:00:05.248755Z","shell.execute_reply":"2022-04-15T17:00:05.253924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_pairs(train):\n    # Calculate all articles purchased together\n    dt = train.groupby(['customer_id','t_dat'])['article_id'].agg(list).rename('pair').reset_index()\n    df = train[['customer_id', 't_dat', 'article_id']].merge(dt, on=['customer_id', 't_dat'], how='left')\n    del dt\n    gc.collect()\n\n    # Explode the rows vs list of articles\n    df = df[['article_id', 'pair']].explode(column='pair')\n    gc.collect()\n    \n    # Discard duplicates\n    df = df.loc[df['article_id']!=df['pair']].reset_index(drop=True)\n    gc.collect()\n\n    # Count how many times each pair combination happens\n    df = df.groupby(['article_id', 'pair']).size().rename('count').reset_index()\n    gc.collect()\n    \n    # Sort by frequency\n    df = df.sort_values(['article_id' ,'count'], ascending=False).reset_index(drop=True)\n    gc.collect()\n    \n    # Pick only top1 most frequent pair\n    df['rank'] = df.groupby('article_id')['pair'].cumcount()\n    df = df.loc[df['rank']==0].reset_index(drop=True)\n    del df['rank']\n    gc.collect()\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:00:05.256492Z","iopub.execute_input":"2022-04-15T17:00:05.256955Z","iopub.status.idle":"2022-04-15T17:00:05.268014Z","shell.execute_reply.started":"2022-04-15T17:00:05.256913Z","shell.execute_reply":"2022-04-15T17:00:05.267069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## cuDF","metadata":{}},{"cell_type":"code","source":"# Load the dataset and discard unused columns\ntrain = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndel train['price']\ndel train['sales_channel_id']\ngc.collect()\n\n# Convert customer_id to int to save memory and speedup processings.\ntrain['customer_id'] = train['customer_id'].factorize()[0].astype('int32')\ntrain['t_dat'] = train['t_dat'].factorize()[0].astype('int16')\ngc.collect()\n\n# number of rows of train\nprint(train.shape)\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:00:06.117788Z","iopub.execute_input":"2022-04-15T17:00:06.118065Z","iopub.status.idle":"2022-04-15T17:00:51.069298Z","shell.execute_reply.started":"2022-04-15T17:00:06.118034Z","shell.execute_reply":"2022-04-15T17:00:51.068459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics_cudf = []\nfor i in range(int(10e5), int(16e6), int(10e5)):\n    times_cudf, _ = mytimeit(calc_pairs, train=train.sample(n=i))\n    metrics_cudf.append(np.mean(times_cudf))","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:00:51.071423Z","iopub.execute_input":"2022-04-15T17:00:51.071953Z","iopub.status.idle":"2022-04-15T17:02:44.890796Z","shell.execute_reply.started":"2022-04-15T17:00:51.071910Z","shell.execute_reply":"2022-04-15T17:02:44.889891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:03:18.961058Z","iopub.execute_input":"2022-04-15T17:03:18.961830Z","iopub.status.idle":"2022-04-15T17:03:18.966784Z","shell.execute_reply.started":"2022-04-15T17:03:18.961780Z","shell.execute_reply":"2022-04-15T17:03:18.966058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## pandas","metadata":{}},{"cell_type":"code","source":"pandas_train = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndel pandas_train['price']\ndel pandas_train['sales_channel_id']\ngc.collect()\n\n# Convert customer_id to int to save memory and speedup processings.\npandas_train['customer_id'] = pandas_train['customer_id'].factorize()[0].astype('int32')\npandas_train['t_dat'] = pandas_train['t_dat'].factorize()[0].astype('int16')\ngc.collect()\n\n# number of rows of train\nprint(pandas_train.shape)\npandas_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:03:28.078124Z","iopub.execute_input":"2022-04-15T17:03:28.078959Z","iopub.status.idle":"2022-04-15T17:04:13.743473Z","shell.execute_reply.started":"2022-04-15T17:03:28.078919Z","shell.execute_reply":"2022-04-15T17:04:13.742689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics_pandas = []\nfor i in range(int(10e5), int(16e6), int(10e5)):\n    times_pd, _ = mytimeit(calc_pairs, train=pandas_train.sample(n=i))\n    metrics_pandas.append(np.mean(times_pd))","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:49:19.918612Z","iopub.execute_input":"2022-04-15T17:49:19.919099Z","iopub.status.idle":"2022-04-15T17:50:59.014338Z","shell.execute_reply.started":"2022-04-15T17:49:19.919054Z","shell.execute_reply":"2022-04-15T17:50:59.013509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Results","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:52:31.314342Z","iopub.execute_input":"2022-04-15T17:52:31.315082Z","iopub.status.idle":"2022-04-15T17:52:31.319586Z","shell.execute_reply.started":"2022-04-15T17:52:31.315024Z","shell.execute_reply":"2022-04-15T17:52:31.318733Z"}}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.rcParams[\"figure.figsize\"] = (30,5)\n\nlengths = list(range(int(10e5), int(16e6), int(10e5)))\nplt.plot(lengths, metrics_pandas, 'o-', label = \"pandas\")\nplt.plot(lengths, metrics_cudf, 'o-', label = \"cuDF\")\nplt.xlabel('DataFrame length')\nplt.ylabel('Execution time(s)')\nplt.title('cuDF vs pandas')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:53:38.462066Z","iopub.execute_input":"2022-04-15T17:53:38.463042Z","iopub.status.idle":"2022-04-15T17:53:38.718672Z","shell.execute_reply.started":"2022-04-15T17:53:38.463005Z","shell.execute_reply":"2022-04-15T17:53:38.718018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'cuDF is {np.mean(metrics_pandas) / np.mean(metrics_cudf)} faster than pandas')","metadata":{"execution":{"iopub.status.busy":"2022-04-15T17:54:36.721691Z","iopub.execute_input":"2022-04-15T17:54:36.722402Z","iopub.status.idle":"2022-04-15T17:54:36.727401Z","shell.execute_reply.started":"2022-04-15T17:54:36.722360Z","shell.execute_reply":"2022-04-15T17:54:36.726358Z"},"trusted":true},"execution_count":null,"outputs":[]}]}