{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**PLEASE UPVOTE https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age**","metadata":{}},{"cell_type":"code","source":"import sys\nimport warnings\nwarnings.filterwarnings('ignore')\nimport time\nimport os\nimport copy\nimport gc\nimport re\nimport random\nimport pickle\nimport cudf\n\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\npd.set_option('display.max_rows', 50)\npd.set_option('display.max_columns', None)\npd.set_option('display.max_colwidth', 10000)\n\nimport seaborn as sns\nsns.set()\n\nfrom pandas.io.json import json_normalize\nfrom pprint import pprint\nfrom pathlib import Path\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom collections import Counter\nfrom datetime import datetime, timedelta","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:48.723927Z","iopub.execute_input":"2022-04-12T10:01:48.724292Z","iopub.status.idle":"2022-04-12T10:01:52.508553Z","shell.execute_reply.started":"2022-04-12T10:01:48.724204Z","shell.execute_reply":"2022-04-12T10:01:52.507741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\nPATH_INPUT = r'../input/h-and-m-personalized-fashion-recommendations/'","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.509868Z","iopub.execute_input":"2022-04-12T10:01:52.510163Z","iopub.status.idle":"2022-04-12T10:01:52.515742Z","shell.execute_reply.started":"2022-04-12T10:01:52.510127Z","shell.execute_reply":"2022-04-12T10:01:52.514768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COUNT = 1\n\nORDER = 0\nN = 12","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.517287Z","iopub.execute_input":"2022-04-12T10:01:52.517736Z","iopub.status.idle":"2022-04-12T10:01:52.525504Z","shell.execute_reply.started":"2022-04-12T10:01:52.517685Z","shell.execute_reply":"2022-04-12T10:01:52.524828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_df(df, head=3):\n    print(f'SHAPE: {df.shape}\\n')\n    display(df.head(head))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.529240Z","iopub.execute_input":"2022-04-12T10:01:52.531016Z","iopub.status.idle":"2022-04-12T10:01:52.535523Z","shell.execute_reply.started":"2022-04-12T10:01:52.530984Z","shell.execute_reply":"2022-04-12T10:01:52.534859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pre_increment(name, local={}):\n    if name in local:\n        local[name] += 1\n        \n        return local[name]\n    \n    globals()[name] += 1\n    \n    return globals()[name]","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.537053Z","iopub.execute_input":"2022-04-12T10:01:52.538047Z","iopub.status.idle":"2022-04-12T10:01:52.544569Z","shell.execute_reply.started":"2022-04-12T10:01:52.537976Z","shell.execute_reply":"2022-04-12T10:01:52.543830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def info_df(df, count, order):\n    if count:\n        try:\n            name = [x for x in globals() if globals()[x] is df][0]\n        except IndexError:\n            name = ''\n        \n        order = pre_increment('order')   \n        print('=' * 30)\n        print(f'{order} INFO_DF {name}:\\n')\n        display_df(df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.545956Z","iopub.execute_input":"2022-04-12T10:01:52.546908Z","iopub.status.idle":"2022-04-12T10:01:52.553869Z","shell.execute_reply.started":"2022-04-12T10:01:52.546868Z","shell.execute_reply":"2022-04-12T10:01:52.553114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df = cudf.read_csv(PATH_INPUT + 'articles.csv', \n                            usecols=['article_id', \n                                     'product_group_name', \n                                     'perceived_colour_master_name'])\ndisplay_df(articles_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:52.555307Z","iopub.execute_input":"2022-04-12T10:01:52.556337Z","iopub.status.idle":"2022-04-12T10:01:54.896946Z","shell.execute_reply.started":"2022-04-12T10:01:52.556253Z","shell.execute_reply":"2022-04-12T10:01:54.896190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df = cudf.read_csv(PATH_INPUT + 'customers.csv',\n                             usecols=['customer_id', 'age'])\ndisplay_df(customers_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:54.898205Z","iopub.execute_input":"2022-04-12T10:01:54.899005Z","iopub.status.idle":"2022-04-12T10:01:57.428234Z","shell.execute_reply.started":"2022-04-12T10:01:54.898966Z","shell.execute_reply":"2022-04-12T10:01:57.427514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df = customers_df.to_pandas()\nbin_list = [-1, 19, 29, 39, 49, 59, 69, 119]\ncustomers_df['age_bins'] = pd.cut(customers_df['age'], bin_list)\n\ndisplay_df(customers_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:57.429715Z","iopub.execute_input":"2022-04-12T10:01:57.430233Z","iopub.status.idle":"2022-04-12T10:01:58.246388Z","shell.execute_reply.started":"2022-04-12T10:01:57.430192Z","shell.execute_reply":"2022-04-12T10:01:58.245707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_missing = customers_df[customers_df['age_bins'].isnull()].shape[0]\nage_missing","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:58.247845Z","iopub.execute_input":"2022-04-12T10:01:58.248099Z","iopub.status.idle":"2022-04-12T10:01:58.265474Z","shell.execute_reply.started":"2022-04-12T10:01:58.248064Z","shell.execute_reply":"2022-04-12T10:01:58.264545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df = cudf.read_csv(PATH_INPUT + 'transactions_train.csv',\n                                usecols=['t_dat', 'customer_id', 'article_id'],\n                                dtype={'t_dat': 'string',\n                                       'customer_id': 'string',\n                                       'article_id': 'int32'})\ntransactions_df['t_dat'] = cudf.to_datetime(transactions_df['t_dat'])\ntransactions_df.set_index('t_dat', inplace=True)\n\ndisplay_df(transactions_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:01:58.267557Z","iopub.execute_input":"2022-04-12T10:01:58.268175Z","iopub.status.idle":"2022-04-12T10:02:35.947887Z","shell.execute_reply.started":"2022-04-12T10:01:58.268124Z","shell.execute_reply":"2022-04-12T10:02:35.947163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recent_df = transactions_df.loc['2020-09-01':'2020-09-21']\n\ndisplay_df(recent_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:35.949355Z","iopub.execute_input":"2022-04-12T10:02:35.949836Z","iopub.status.idle":"2022-04-12T10:02:37.245560Z","shell.execute_reply.started":"2022-04-12T10:02:35.949798Z","shell.execute_reply":"2022-04-12T10:02:37.244320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recent_df = recent_df.to_pandas()\nrecent_df = recent_df.merge(customers_df[['customer_id', 'age_bins']], on='customer_id', how='inner')\n\ndisplay_df(recent_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:37.249528Z","iopub.execute_input":"2022-04-12T10:02:37.249760Z","iopub.status.idle":"2022-04-12T10:02:38.304559Z","shell.execute_reply.started":"2022-04-12T10:02:37.249733Z","shell.execute_reply":"2022-04-12T10:02:38.303760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recent_df = recent_df.groupby(['age_bins', 'article_id']).count().reset_index()\\\n    .rename(columns={'customer_id': 'counts'})\n\ndisplay_df(recent_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:38.305782Z","iopub.execute_input":"2022-04-12T10:02:38.306091Z","iopub.status.idle":"2022-04-12T10:02:40.203427Z","shell.execute_reply.started":"2022-04-12T10:02:38.306042Z","shell.execute_reply":"2022-04-12T10:02:40.202608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_unique_list = recent_df['age_bins'].unique().tolist()\nbins_unique_list","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.204684Z","iopub.execute_input":"2022-04-12T10:02:40.205134Z","iopub.status.idle":"2022-04-12T10:02:40.215730Z","shell.execute_reply.started":"2022-04-12T10:02:40.205092Z","shell.execute_reply":"2022-04-12T10:02:40.214768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_unique_list[0]","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.217337Z","iopub.execute_input":"2022-04-12T10:02:40.217661Z","iopub.status.idle":"2022-04-12T10:02:40.225159Z","shell.execute_reply.started":"2022-04-12T10:02:40.217622Z","shell.execute_reply":"2022-04-12T10:02:40.223234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ages_article_id = {}\ncount = COUNT\norder = ORDER\nfor bins_unique in bins_unique_list:\n    temp_df = recent_df[recent_df['age_bins'] == bins_unique]\n    info_df(temp_df, count,order)\n    \n    temp_df = temp_df.sort_values(by='counts', ascending=False)\n    info_df(temp_df, count, order)\n    \n    ages_article_id[bins_unique] = temp_df.head(100)['article_id'].values.tolist()\n    \n    count = 0","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.226821Z","iopub.execute_input":"2022-04-12T10:02:40.227608Z","iopub.status.idle":"2022-04-12T10:02:40.269954Z","shell.execute_reply.started":"2022-04-12T10:02:40.227570Z","shell.execute_reply":"2022-04-12T10:02:40.269334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import islice\n\nfor key, value in islice(ages_article_id.items(), 3):\n    print(f'{key} {len(value)}')","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.270950Z","iopub.execute_input":"2022-04-12T10:02:40.271199Z","iopub.status.idle":"2022-04-12T10:02:40.276738Z","shell.execute_reply.started":"2022-04-12T10:02:40.271164Z","shell.execute_reply":"2022-04-12T10:02:40.275845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topcustcnt_byage_df = pd.DataFrame([ages_article_id])\n\ndisplay_df(topcustcnt_byage_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.278398Z","iopub.execute_input":"2022-04-12T10:02:40.279147Z","iopub.status.idle":"2022-04-12T10:02:40.305662Z","shell.execute_reply.started":"2022-04-12T10:02:40.279108Z","shell.execute_reply":"2022-04-12T10:02:40.304845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topcustcnt_byage_df = pd.DataFrame([ages_article_id]).T.rename(columns={0: 'top_100'})\n\ndisplay_df(topcustcnt_byage_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.306964Z","iopub.execute_input":"2022-04-12T10:02:40.307283Z","iopub.status.idle":"2022-04-12T10:02:40.323371Z","shell.execute_reply.started":"2022-04-12T10:02:40.307248Z","shell.execute_reply":"2022-04-12T10:02:40.322488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in topcustcnt_byage_df.index:\n    topcustcnt_byage_df[i] = [len(set(topcustcnt_byage_df.at[i, 'top_100']) & \\\n                                  set(topcustcnt_byage_df.at[j, 'top_100'])) / 100 for j in topcustcnt_byage_df.index]\n    \ndisplay_df(topcustcnt_byage_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.325292Z","iopub.execute_input":"2022-04-12T10:02:40.325914Z","iopub.status.idle":"2022-04-12T10:02:40.369032Z","shell.execute_reply.started":"2022-04-12T10:02:40.325875Z","shell.execute_reply":"2022-04-12T10:02:40.368308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topcustcnt_byage_df = topcustcnt_byage_df.drop(columns='top_100')\n\ndisplay_df(topcustcnt_byage_df, head=10)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.370342Z","iopub.execute_input":"2022-04-12T10:02:40.370810Z","iopub.status.idle":"2022-04-12T10:02:40.390577Z","shell.execute_reply.started":"2022-04-12T10:02:40.370758Z","shell.execute_reply":"2022-04-12T10:02:40.389840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.heatmap(topcustcnt_byage_df, cmap='winter', annot=True, cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.391829Z","iopub.execute_input":"2022-04-12T10:02:40.392086Z","iopub.status.idle":"2022-04-12T10:02:40.836392Z","shell.execute_reply.started":"2022-04-12T10:02:40.392053Z","shell.execute_reply":"2022-04-12T10:02:40.835613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_unique_list","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.838001Z","iopub.execute_input":"2022-04-12T10:02:40.838500Z","iopub.status.idle":"2022-04-12T10:02:40.844703Z","shell.execute_reply.started":"2022-04-12T10:02:40.838460Z","shell.execute_reply":"2022-04-12T10:02:40.844007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_unique_list = customers_df['age_bins'].unique().tolist()\nbins_unique_list","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.845965Z","iopub.execute_input":"2022-04-12T10:02:40.846451Z","iopub.status.idle":"2022-04-12T10:02:40.864772Z","shell.execute_reply.started":"2022-04-12T10:02:40.846411Z","shell.execute_reply":"2022-04-12T10:02:40.864106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = COUNT\norder = ORDER\nfor bin_unique in bins_unique_list:\n    \n    df = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n                       usecols=['t_dat', 'customer_id', 'article_id'],\n                       dtype={'t_dat': 'string',\n                              'customer_id': 'string',\n                              'article_id': 'int32'})\n    info_df(df, count, order)\n    \n    if str(bin_unique) == 'nan':\n        temp_customers_df = customers_df[customers_df['age_bins'].isnull()]\n    else:\n        temp_customers_df = customers_df[customers_df['age_bins'] == bin_unique]\n        \n    temp_customers_df = temp_customers_df.drop(['age_bins'], axis=1)\n    temp_customers_df = cudf.from_pandas(temp_customers_df)\n    info_df(temp_customers_df, count, order)\n    \n    df = df.merge(temp_customers_df[['customer_id', 'age']], on='customer_id', how='inner')\n    info_df(df, count, order)\n    \n    print(f'TRANSACTION SHAPE FOR SCOPE {bin_unique}: {df.shape}\\n')\n    \n    df['customer_id'] = df['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n    df['t_dat'] = cudf.to_datetime(df['t_dat'])\n    \n    last_date_train = df['t_dat'].max()\n    \n    if count:\n        print('=' * 30 + f'\\n4 LAST_DATE_TRAIN:\\n{last_date_train}')\n    \n    tmp = df[['t_dat']].copy().to_pandas()\n    tmp['weekday'] = tmp['t_dat'].dt.dayofweek\n    info_df(tmp, count, order)\n    \n    tmp['t_dat_shiftday'] = tmp['t_dat'] - pd.TimedeltaIndex(tmp['weekday'] - 1, unit='D')\n    info_df(tmp, count, order)\n    \n    tmp.loc[tmp['weekday'] >= 2, 't_dat_shiftday'] = tmp.loc[tmp['weekday'] >= 2, 't_dat_shiftday'] + \\\n        pd.TimedeltaIndex(np.ones(len(tmp.loc[tmp['weekday'] >= 2])) * 7, unit='D')\n    info_df(tmp, count, order)\n    \n    df['t_dat_shiftday'] = tmp['t_dat_shiftday'].values\n    info_df(df, count, order)\n    \n    weekly_sales = df.drop('customer_id', axis=1).groupby(['t_dat_shiftday', 'article_id']).count().reset_index()\n    info_df(weekly_sales, count, order)\n    \n    weekly_sales = weekly_sales.rename(columns={'t_dat': 'count'})\n    info_df(weekly_sales, count, order)\n    \n    df = df.merge(weekly_sales, on=['t_dat_shiftday', 'article_id'], how='left')\n    info_df(df, count, order)\n    \n    weekly_sales = weekly_sales.reset_index().set_index('article_id')\n    info_df(weekly_sales, count, order)\n    \n    df = df.merge(weekly_sales.loc[weekly_sales['t_dat_shiftday'] == last_date_train, ['count']], \n                 on='article_id',\n                 suffixes=('', '_targ'))\n    info_df(df, count, order)\n    \n    df['count_targ'].fillna(0, inplace=True)\n    del weekly_sales\n    \n    df['quotient'] = df['count_targ'] / df['count']\n    info_df(df, count, order)\n    \n    target_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\n    info_df(target_sales, count, order)\n    \n    general_pred = target_sales.nlargest(N).index.to_pandas().tolist()\n    general_pred = ['0' + str(article_id) for article_id in general_pred]\n    general_pred_str = ' '.join(general_pred)\n    \n    del target_sales\n    \n    purchase_dict = {}\n    tmp = df.copy().to_pandas()\n    info_df(tmp, count, order)\n    \n    tmp['x'] = ((last_date_train - tmp['t_dat']) / np.timedelta64(1, 'D')).astype(int)\n    info_df(tmp, count, order)\n    \n    tmp['dummy_1'] = 1\n    tmp['x'] = tmp[['x', 'dummy_1']].max(axis=1)\n    info_df(tmp, count, order)\n    \n    a, b, c, d = 2.5e4, 1.5e5, 2e-1, 1e3\n    tmp['y'] = a / np.sqrt(tmp['x']) + b * np.exp(-c * tmp['x']) - d\n    info_df(tmp, count, order)\n    \n    tmp['dummy_0'] = 0\n    tmp['y'] = tmp[['y', 'dummy_0']].max(axis=1)\n    tmp['value'] = tmp['quotient'] * tmp['y']\n    info_df(tmp, count, order)\n    \n    tmp = tmp.groupby(['customer_id', 'article_id']).agg({'value': 'sum'})\n    info_df(tmp, count, order)\n    \n    tmp = tmp.reset_index()\n    \n    tmp = tmp.loc[tmp['value'] > 0]\n    info_df(tmp, count, order)\n    \n    tmp['rank'] = tmp.groupby('customer_id')['value'].rank('dense', ascending=False)\n    info_df(tmp, count, order)\n    \n    tmp = tmp.loc[tmp['rank'] <= 12]\n    info_df(tmp, count, order)\n    \n    purchase_df = tmp.sort_values(['customer_id', 'value'], ascending=False).reset_index(drop=True)\n    info_df(purchase_df, count, order)\n    \n    purchase_df['prediction'] = '0' + purchase_df['article_id'].astype('str') + ' '\n    info_df(purchase_df, count, order)\n    \n    purchase_df = purchase_df.groupby('customer_id').agg({'prediction': sum}).reset_index()\n    info_df(purchase_df, count, order)\n    \n    purchase_df = cudf.DataFrame(purchase_df)\n    \n    sub = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv',\n                        usecols=['customer_id'],\n                        dtype={'customer_id': 'string'})\n    info_df(sub, count, order)\n    \n    num_customers = sub.shape[0]\n    sub = sub.merge(temp_customers_df[['customer_id', 'age']], on='customer_id', how='inner')\n    info_df(sub, count, order)\n    \n    sub['customer_id_2'] = sub['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n    info_df(sub, count, order)\n    \n    sub = sub.merge(purchase_df,\n                    left_on='customer_id_2',\n                    right_on='customer_id',\n                    how='left',\n                    suffixes={'', '_ignored'})\n    info_df(sub, count, order)\n    \n    sub = sub.to_pandas()\n    sub['prediction'] = sub['prediction'].fillna(general_pred_str)\n    sub['prediction'] = sub['prediction'] + ' ' + general_pred_str\n    sub['prediction'] = sub['prediction'].str.strip()\n    sub['prediction'] = sub['prediction'].str[:131]\n    \n    sub = sub[['customer_id', 'prediction']]\n    sub.to_csv(f'submission_{str(bin_unique)}.csv', index=False)\n    info_df(sub, count, order)\n    \n    print(f'SAVED PREDICTION FOR {bin_unique}, SHAPE: {sub.shape}\\n')\n    print('-' * 50)\n    \n    count = 0\n    \nprint('FINISHED')\nprint('=' * 50)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:02:40.866087Z","iopub.execute_input":"2022-04-12T10:02:40.866581Z","iopub.status.idle":"2022-04-12T10:03:45.991829Z","shell.execute_reply.started":"2022-04-12T10:02:40.866546Z","shell.execute_reply":"2022-04-12T10:03:45.991082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, bin_unique in enumerate(bins_unique_list):\n    temp_df = cudf.read_csv(f'submission_{str(bin_unique)}.csv')\n    if i == 0:\n        sub_df = temp_df\n    else:\n        sub_df = cudf.concat([sub_df, temp_df], axis=0)\n        \nassert sub_df.shape[0] == num_customers,\\\n    f'SUB_DF ROWS NUMBER IS NOT CORRECT {sub_df.shape[0]} VS {num_customers}'\n    \nsub_df.to_csv('by_cust_age__.csv', index=False)\n\ndisplay_df(sub_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:03:45.993162Z","iopub.execute_input":"2022-04-12T10:03:45.993804Z","iopub.status.idle":"2022-04-12T10:03:47.152657Z","shell.execute_reply.started":"2022-04-12T10:03:45.993748Z","shell.execute_reply":"2022-04-12T10:03:47.151852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_df = cudf.read_csv('./by_cust_age__.csv')\n\ndisplay_df(check_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T10:03:47.153928Z","iopub.execute_input":"2022-04-12T10:03:47.154729Z","iopub.status.idle":"2022-04-12T10:03:47.363997Z","shell.execute_reply.started":"2022-04-12T10:03:47.154688Z","shell.execute_reply":"2022-04-12T10:03:47.363162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}