{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"font-family:verdana;\"><center>🗂️OTTO: LightGBM brief sample code with cudf</center></h1>","metadata":{}},{"cell_type":"markdown","source":"# References\n**Great notebooks! Thank you🙇**\n- https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-575\n- https://www.kaggle.com/code/greenwolf/lightgbm-fast-recall-20/notebook","metadata":{}},{"cell_type":"markdown","source":"**My other notebook📒:**\n- 🥉https://www.kaggle.com/code/mizuny/otto-easy-to-use-co-visitation-matrix-function\n- https://www.kaggle.com/code/mizuny/otto-eda-notebook","metadata":{}},{"cell_type":"markdown","source":"# In this notebook\n- LightGBM ranking training & validation\n- Fast reading data with cudf\n- Brief commentary for LightGBM parameter `group`","metadata":{}},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, gc\nimport cudf\nprint('We will use RAPIDS version',cudf.__version__)\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:22:24.198004Z","iopub.execute_input":"2023-01-29T07:22:24.198783Z","iopub.status.idle":"2023-01-29T07:22:28.926453Z","shell.execute_reply.started":"2023-01-29T07:22:24.198690Z","shell.execute_reply":"2023-01-29T07:22:28.925239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:14px; font-family:verdana;\">\n    📌For description of loading data, read following notebook: <a href=\"https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-575\" target=\"_blank\" style=\"color:#ff7f7f;\">Candidate ReRank Model - [LB 0.575] by cdeotte</a>\n</div>","metadata":{}},{"cell_type":"code","source":"%%time\n# CACHE FUNCTIONS\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\n\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)\n\n# CHUNK PARAMETERS\nREAD_CT = 5\nOUTER_CHUNK_NUM = 100\nCHUNK_NUM = int( np.ceil( len(files) / OUTER_CHUNK_NUM ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK_NUM}.')","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:22:28.928566Z","iopub.execute_input":"2023-01-29T07:22:28.928982Z","iopub.status.idle":"2023-01-29T07:23:29.792452Z","shell.execute_reply.started":"2023-01-29T07:22:28.928943Z","shell.execute_reply":"2023-01-29T07:23:29.790986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Data","metadata":{}},{"cell_type":"code","source":"drop_th_sess_num = 30\n\nfor outer_chk_id in range(OUTER_CHUNK_NUM):\n    start_chk_id = outer_chk_id * CHUNK_NUM\n    end_chk_id = min( (outer_chk_id+1)*CHUNK_NUM, len(files) )\n\n    for inner_chk_id in range(start_chk_id, end_chk_id, READ_CT):\n        df = [read_file(files[inner_chk_id])]\n        # read inner chunk files\n        for i in range(1, READ_CT): \n            if (inner_chk_id+i) < end_chk_id:\n                df.append( read_file(files[(inner_chk_id+i)]) )\n\n        df = cudf.concat(df,ignore_index=True,axis=0)\n\n        # USE TAIL OF SESSION\n        df = df.reset_index(drop=True)\n        df['n'] = df.groupby('session').cumcount()\n        # ** SPECIFY BY ARG **\n        df = df.loc[df.n < drop_th_sess_num].drop('n',axis=1)\n        \n        # COMBINE INNER CHUNKS\n        if inner_chk_id == start_chk_id:\n            df_concat = df\n        else:\n            df_concat = cudf.concat([df_concat, df], ignore_index=True, axis=0)\n        print(inner_chk_id,', ',end='')\n\n    del df\n    gc.collect()\n    break # if use all data. please comment out","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:29.794255Z","iopub.execute_input":"2023-01-29T07:23:29.794937Z","iopub.status.idle":"2023-01-29T07:23:32.045019Z","shell.execute_reply.started":"2023-01-29T07:23:29.794894Z","shell.execute_reply":"2023-01-29T07:23:32.043893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_concat = df_concat.sort_values('ts',ascending=False)\ndf_concat = df_concat.reset_index(drop=True)\ndf_concat","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.048400Z","iopub.execute_input":"2023-01-29T07:23:32.048707Z","iopub.status.idle":"2023-01-29T07:23:32.119879Z","shell.execute_reply.started":"2023-01-29T07:23:32.048677Z","shell.execute_reply":"2023-01-29T07:23:32.118716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify boundary index between train and val\n# 80% for train, 20% for val\nfold_size = len(df_concat) // 5","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.121882Z","iopub.execute_input":"2023-01-29T07:23:32.122351Z","iopub.status.idle":"2023-01-29T07:23:32.127740Z","shell.execute_reply.started":"2023-01-29T07:23:32.122299Z","shell.execute_reply":"2023-01-29T07:23:32.126481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create train / val dataset","metadata":{}},{"cell_type":"code","source":"val_df = df_concat.iloc[:fold_size]\nval_df = val_df.reset_index(drop=True)\n\ntrain_df = df_concat.iloc[fold_size:]\ntrain_df = train_df.reset_index(drop=True)\n\ndel df_concat\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.129365Z","iopub.execute_input":"2023-01-29T07:23:32.130096Z","iopub.status.idle":"2023-01-29T07:23:32.281651Z","shell.execute_reply.started":"2023-01-29T07:23:32.130057Z","shell.execute_reply":"2023-01-29T07:23:32.280462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_query = train_df.groupby(by='session', sort=True).size()\nval_query = val_df.groupby(by='session', sort=True).size()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.283303Z","iopub.execute_input":"2023-01-29T07:23:32.283767Z","iopub.status.idle":"2023-01-29T07:23:32.316986Z","shell.execute_reply.started":"2023-01-29T07:23:32.283729Z","shell.execute_reply":"2023-01-29T07:23:32.316156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **`group` parameter for LightGBM**\n> group (list, numpy 1-D array, pandas Series or None, optional (default=None)) – Group/query data. Only used in the learning-to-rank task. sum(group) = n_samples. For example, if you have a 100-document dataset with group = [10, 20, 40, 10, 10, 10], that means that you have 6 groups, where the first 10 records are in the first group, records 11-30 are in the second group, records 31-70 are in the third group, etc.\nby [LightGBM documentation](https://lightgbm.readthedocs.io/en/latest/pythonapi/lightgbm.Dataset.html)","metadata":{}},{"cell_type":"code","source":"print(f'train size: {len(train_df)}, ')\nprint(f'val size: {len(val_df)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.318638Z","iopub.execute_input":"2023-01-29T07:23:32.319013Z","iopub.status.idle":"2023-01-29T07:23:32.325708Z","shell.execute_reply.started":"2023-01-29T07:23:32.318978Z","shell.execute_reply":"2023-01-29T07:23:32.323658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_col = 'type'\ntrain_dataset = lgb.Dataset(\n    train_df.drop([target_col], axis=1).values.get(),\n    train_df[target_col].values.astype(int).get(),\n    group=train_query.values.get())\n\nval_dataset = lgb.Dataset(\n    val_df.drop([target_col], axis=1).values.get(),\n    val_df[target_col].values.astype(int).get(),\n    group=val_query.values.get())","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:23:32.328193Z","iopub.execute_input":"2023-01-29T07:23:32.329055Z","iopub.status.idle":"2023-01-29T07:23:33.445014Z","shell.execute_reply.started":"2023-01-29T07:23:32.329019Z","shell.execute_reply":"2023-01-29T07:23:33.444102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nparams = {\n    'objective': 'lambdarank',\n#     'metric': '\"None\"', # if use custom metric, specify \"None\"\n    'learning_rate': 0.01,\n    'boosting_type': 'gbdt',\n#     'label_gain': label_gain\n}\n\nmodel = lgb.train(\n    params,\n    train_dataset,\n    valid_sets=val_dataset,\n#     feval=custom_metric_func,\n    callbacks=[lgb.log_evaluation(5)],\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:24:02.732543Z","iopub.execute_input":"2023-01-29T07:24:02.732909Z","iopub.status.idle":"2023-01-29T07:24:09.936829Z","shell.execute_reply.started":"2023-01-29T07:24:02.732876Z","shell.execute_reply":"2023-01-29T07:24:09.936041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(val_df.drop(['type'], axis=1).values.get())","metadata":{"execution":{"iopub.status.busy":"2023-01-29T07:24:19.403571Z","iopub.execute_input":"2023-01-29T07:24:19.403940Z","iopub.status.idle":"2023-01-29T07:24:19.752751Z","shell.execute_reply.started":"2023-01-29T07:24:19.403906Z","shell.execute_reply":"2023-01-29T07:24:19.751952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}