{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:03.753359Z","iopub.execute_input":"2023-01-30T21:35:03.753626Z","iopub.status.idle":"2023-01-30T21:35:10.689661Z","shell.execute_reply.started":"2023-01-30T21:35:03.753600Z","shell.execute_reply":"2023-01-30T21:35:10.688460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pandarallel","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:10.691640Z","iopub.execute_input":"2023-01-30T21:35:10.691981Z","iopub.status.idle":"2023-01-30T21:35:15.460576Z","shell.execute_reply.started":"2023-01-30T21:35:10.691947Z","shell.execute_reply":"2023-01-30T21:35:15.459644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install ipywidgets","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:15.461897Z","iopub.execute_input":"2023-01-30T21:35:15.462229Z","iopub.status.idle":"2023-01-30T21:35:20.071224Z","shell.execute_reply.started":"2023-01-30T21:35:15.462192Z","shell.execute_reply":"2023-01-30T21:35:20.070242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install lightgbm","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:20.073764Z","iopub.execute_input":"2023-01-30T21:35:20.074159Z","iopub.status.idle":"2023-01-30T21:35:27.070944Z","shell.execute_reply.started":"2023-01-30T21:35:20.074129Z","shell.execute_reply":"2023-01-30T21:35:27.069986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install polars","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:27.072469Z","iopub.execute_input":"2023-01-30T21:35:27.073214Z","iopub.status.idle":"2023-01-30T21:35:32.171932Z","shell.execute_reply.started":"2023-01-30T21:35:27.073172Z","shell.execute_reply":"2023-01-30T21:35:32.170993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install daal4py'>=2020.3'","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:32.173329Z","iopub.execute_input":"2023-01-30T21:35:32.173662Z","iopub.status.idle":"2023-01-30T21:35:43.097189Z","shell.execute_reply.started":"2023-01-30T21:35:32.173629Z","shell.execute_reply":"2023-01-30T21:35:43.096171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VER = 6\nimport daal4py as d4p\nimport lightgbm as lgb\nimport pandas as pd, numpy as np\nfrom tqdm import tqdm\ntqdm.pandas()\n\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport itertools\n\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\n\nfrom pandarallel import pandarallel\n\npandarallel.initialize(nb_workers=24, progress_bar=True, use_memory_fs=False)\n\nimport polars as pl\nfrom pyarrow.parquet import ParquetFile\nimport pyarrow as pa\nimport re","metadata":{"papermill":{"duration":2.696152,"end_time":"2022-11-28T18:49:28.3499","exception":false,"start_time":"2022-11-28T18:49:25.653748","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-30T21:35:43.098674Z","iopub.execute_input":"2023-01-30T21:35:43.099373Z","iopub.status.idle":"2023-01-30T21:35:47.486917Z","shell.execute_reply.started":"2023-01-30T21:35:43.099334Z","shell.execute_reply":"2023-01-30T21:35:47.485869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GENERATE_FOR = \"kaggle\" # \"kaggle\"\nRUN_FOR = \"kaggle\" # \"kaggle\"\ntype_labels = {'clicks':0,'carts':1, 'orders':2}\nCANDIDATE_COUNT = 100","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:47.488248Z","iopub.execute_input":"2023-01-30T21:35:47.488675Z","iopub.status.idle":"2023-01-30T21:35:47.493320Z","shell.execute_reply.started":"2023-01-30T21:35:47.488647Z","shell.execute_reply":"2023-01-30T21:35:47.492504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if GENERATE_FOR == \"local\":\n    train_path = \"/kaggle/input/otto-generating-splits/train.parquet\"\n    val_path = \"/kaggle/input/otto-generating-splits/val.parquet\"\n    \nelif GENERATE_FOR == \"kaggle\":\n    train_path = \"/kaggle/input/otto-generating-splits/all_train.parquet\"\n    val_path = \"/kaggle/input/otto-generating-splits/test.parquet\"","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:47.494458Z","iopub.execute_input":"2023-01-30T21:35:47.494784Z","iopub.status.idle":"2023-01-30T21:35:47.507797Z","shell.execute_reply.started":"2023-01-30T21:35:47.494756Z","shell.execute_reply":"2023-01-30T21:35:47.506968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{"execution":{"iopub.status.busy":"2023-01-10T18:49:19.234587Z","iopub.execute_input":"2023-01-10T18:49:19.235643Z","iopub.status.idle":"2023-01-10T18:49:19.258644Z","shell.execute_reply.started":"2023-01-10T18:49:19.235598Z","shell.execute_reply":"2023-01-10T18:49:19.257203Z"}}},{"cell_type":"code","source":"train_sessions = np.load(\"/kaggle/input/otto-generating-splits/val_sessions_for_train.npy\", allow_pickle=True)\nitem_features = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_item_features.pqt')\nuser_features = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_user_features.pqt')\nuser_item_int_features = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_user_item_int_features.pqt')\nall_clicks_covisit_feature_df = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_all_clicks_covisitation_features.pqt')\nall_cart_covisit_feature_df = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_all_carts_orders_covisitation_features.pqt')\nall_buy2buy_covisit_feature_df = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_all_buy2buy_covisitation_features.pqt')\naid_occurences = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_aid_occurences.pqt')\n#aid_vectors = pd.read_parquet('/kaggle/input/otto-word2vec-embedding-features-kaggle-50/aid_vectors_kaggle.pqt')\n#aid_vectors = pd.DataFrame(aid_vectors.vectors.tolist(),columns = [\"vector_\"+str(col) for col in range(1,33)],index=aid_vectors.aid).reset_index()\n#aid_vectors = pl.from_pandas(aid_vectors)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:35:47.510063Z","iopub.execute_input":"2023-01-30T21:35:47.510332Z","iopub.status.idle":"2023-01-30T21:35:47.657774Z","shell.execute_reply.started":"2023-01-30T21:35:47.510310Z","shell.execute_reply":"2023-01-30T21:35:47.656886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nsubs = []\n\nfor type_str in tqdm(list(type_labels.keys())):\n    pf = ParquetFile(f'/kaggle/input/2-generation-candidates-{RUN_FOR}-{CANDIDATE_COUNT}/{RUN_FOR}_{CANDIDATE_COUNT}_candidates_{type_str}.parquet')\n    print('Reading',type_str,'data')\n    covisit_feature_df = pl.scan_parquet(f'/kaggle/input/3-feature-extraction-{GENERATE_FOR}-{CANDIDATE_COUNT}-v3/{GENERATE_FOR}_covisitation_features_{type_str}_{CANDIDATE_COUNT}_candidates.pqt')\n    \n    chunk = round(pf.metadata.num_rows/2)\n    print(chunk)\n    model_paths = sorted(glob.glob(f\"/kaggle/input/4-train-v3/LGB_{CANDIDATE_COUNT}candidates_fold*_{type_str}.txt\"))\n\n    all_predictions = []\n\n    for batch_i, batch in tqdm(enumerate(pf.iter_batches(batch_size = chunk))):\n        whole_df = batch.to_pandas()\n        whole_df = pl.from_pandas(whole_df).explode(\"aid\")\n        #candidate_df = pl.read_parquet(f\"/kaggle/input/2-generation-candidates-{GENERATE_FOR}/{GENERATE_FOR}_{CANDIDATE_COUNT}_candidates_{type_str}.parquet\").explode(\"aid\")\n        print(\"Candidate df shape:\", whole_df.shape)\n        rank_repeater = np.hstack([list(range(1,CANDIDATE_COUNT+1)) for i in range(int(len(whole_df)/CANDIDATE_COUNT))])\n        whole_df = whole_df.with_column(pl.Series(name=\"candidate_rank\", values=rank_repeater))\n        print('Candidate Rank Features Merge, Done.')\n        del rank_repeater;gc.collect()\n        if RUN_FOR == \"local\":\n            whole_df = whole_df.to_pandas()\n            whole_df = whole_df[~whole_df.session.isin(train_sessions)].reset_index(drop=True)\n            whole_df = pl.from_pandas(whole_df)\n\n        #Merging Features\n        whole_df = whole_df.join(covisit_feature_df, on=['session','aid'], how='left').fill_null(-1)\n        print('Covisit Features 1, Done.')\n        whole_df = whole_df.unique()\n        \n        whole_df = whole_df.join(item_features, on='aid', how='left').fill_null(-1)\n        whole_df = whole_df.join(aid_occurences, on='aid', how='left').fill_null(-1)\n        print('Item Features Merge, Done.')\n        whole_df = whole_df.join(user_features, on='session', how='left').fill_null(-1)\n        print('User Features Merge, Done.')\n        whole_df = whole_df.join(user_item_int_features,\n                                              on=['session', 'aid'],\n                                              how='left').fill_null(-1)\n        print('User-Item Features, Done.')\n        \n        whole_df = whole_df.join(all_clicks_covisit_feature_df, on=['aid'], how='left').fill_null(-1)\n        whole_df = whole_df.join(all_cart_covisit_feature_df, on=['aid'], how='left').fill_null(-1)\n        whole_df = whole_df.join(all_buy2buy_covisit_feature_df, on=['aid'], how='left').fill_null(-1)\n        print('Covisit Features 2, Done.')\n        #whole_df = whole_df.join(aid_vectors, on=['aid'], how='left').fill_null(-999)\n        #print('w2v Features 2, Done.')\n        whole_df = whole_df.to_pandas()\n        whole_df = whole_df.rename(columns = lambda x:re.sub('[^A-Za-z0-9_]+', '', x))\n        FEATURES = whole_df.columns[2:]\n        FEATURES = [col for col in FEATURES if 'index' not in col]\n        #FEATURES = whole_df.columns[2:]\n        print(\"Feature count:\",len(FEATURES))\n        preds = np.zeros(len(whole_df))\n        \n        for model_path in model_paths:\n            print(\"Model:\",model_path)\n            model = lgb.Booster(model_file=model_path)\n            daal_model = d4p.get_gbt_model_from_lightgbm(model)\n            pred = d4p.gbt_regression_prediction().compute(whole_df[FEATURES], daal_model).prediction\n            preds += pred.reshape(1,-1)[0]/len(model_paths)\n            #preds += model.predict(whole_df[FEATURES])/len(model_paths)\n        predictions = whole_df[['session','aid']].copy()\n        predictions['pred'] = preds\n        all_predictions.append(predictions)\n        \n    all_predictions = pd.concat(all_predictions, ignore_index=True)\n    all_predictions['type'] = type_str\n    all_predictions = all_predictions.sort_values(['session','pred'],\n                                                  ascending=[True,False]).reset_index(drop=True)\n    #all_predictions['n'] = all_predictions.groupby('session').aid.cumcount().astype('int8')\n    #all_predictions = all_predictions.loc[all_predictions.n<20]\n    #all_predictions = pl.from_pandas(all_predictions).groupby(\"session\").head(20).to_pandas().reset_index(drop=True)\n    #sub = all_predictions.groupby('session').aid.apply(list)\n    #sub = sub.to_frame().reset_index()\n    #sub.item = sub.aid.apply(lambda x: \" \".join(map(str,x)))\n    #sub.columns = ['session_type','labels']\n    #sub['session_type'] = sub.session_type.astype('str') + '_' + type_str\n    #print(\"submission shape:\",sub.shape)\n    subs.append(all_predictions)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T21:36:13.423036Z","iopub.execute_input":"2023-01-30T21:36:13.423512Z","iopub.status.idle":"2023-01-30T22:37:18.455538Z","shell.execute_reply.started":"2023-01-30T21:36:13.423477Z","shell.execute_reply":"2023-01-30T22:37:18.454568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wall time: 46min 11s","metadata":{}},{"cell_type":"markdown","source":"## Local CV/Submission","metadata":{}},{"cell_type":"code","source":"final_sub = pd.concat(subs, ignore_index=True)\nfinal_sub.to_parquet('sub_all.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-01-30T22:37:18.457395Z","iopub.execute_input":"2023-01-30T22:37:18.457756Z","iopub.status.idle":"2023-01-30T22:38:46.586758Z","shell.execute_reply.started":"2023-01-30T22:37:18.457696Z","shell.execute_reply":"2023-01-30T22:38:46.585153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub","metadata":{"execution":{"iopub.status.busy":"2023-01-30T22:38:46.617489Z","iopub.execute_input":"2023-01-30T22:38:46.617787Z","iopub.status.idle":"2023-01-30T22:38:46.630543Z","shell.execute_reply.started":"2023-01-30T22:38:46.617761Z","shell.execute_reply":"2023-01-30T22:38:46.629811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub.session.value_counts().sort_values()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T22:38:48.310341Z","iopub.execute_input":"2023-01-30T22:38:48.310653Z","iopub.status.idle":"2023-01-30T22:38:55.944081Z","shell.execute_reply.started":"2023-01-30T22:38:48.310629Z","shell.execute_reply":"2023-01-30T22:38:55.943144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"final_sub = pd.concat(subs, ignore_index=True)\nfinal_sub.sort_values(by=\"session_type\", ascending=True).reset_index(drop=True)\nfinal_sub.to_parquet('final_sub.parquet')\nif RUN_FOR == \"local\":\n    # COMPUTE METRIC\n    score = 0\n    weights = {'clicks': 0.10, 'carts': 0.30, 'orders': 0.60}\n    for t in ['clicks','carts','orders']:\n        sub = final_sub.loc[final_sub.session_type.str.contains(t)].copy()\n        sub['session'] = sub.session_type.apply(lambda x: int(x.split('_')[0]))\n        test_labels = pd.read_parquet('/kaggle/input/otto-generating-splits/val_labels.parquet')\n        test_labels = test_labels[~test_labels.session.isin(train_sessions)].reset_index(drop=True)\n        test_labels = test_labels.loc[test_labels['type']==t]\n        test_labels = test_labels.merge(sub, how='left', on=['session'])\n        test_labels['hits'] = test_labels.apply(lambda df: len(set(df.ground_truth).intersection(set(df.labels))), axis=1)\n        test_labels['gt_count'] = test_labels.ground_truth.str.len().clip(0,20)\n        recall = test_labels['hits'].sum() / test_labels['gt_count'].sum()\n        score += weights[t]*recall\n        print(f'{t} recall =',recall)\n\n    print('=============')\n    print('Overall Recall =',score)\n    print('=============')\n\nelif RUN_FOR == \"kaggle\":\n    final_sub[\"labels\"] = final_sub.labels.apply(lambda x: \" \".join([str(elm) for elm in x]))\n    final_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T13:32:45.010780Z","iopub.execute_input":"2023-01-23T13:32:45.011085Z","iopub.status.idle":"2023-01-23T13:34:21.551003Z","shell.execute_reply.started":"2023-01-23T13:32:45.011059Z","shell.execute_reply":"2023-01-23T13:34:21.549851Z"}}},{"cell_type":"markdown","source":"Local:\n\nclicks recall = 0.54130354363016\ncarts recall = 0.426388098296038\norders recall = 0.6612103200694317\n=============\nOverall Recall = 0.5787729758934864\n=============","metadata":{}}]}