{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n\n**I am inspired from [RADEK OSMULSK](https://www.kaggle.com/radek1)'s kernel : [co-visitation matrix](https://www.kaggle.com/radek1/code).**\n\n**In this kernel, I try 'Re-Rank History Item' Architecture.**","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ[\"PYTHONIOENCODING\"] = \"utf8\"\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\n\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nimport random as rnd\nimport shutil\nimport gc\nimport datetime\nfrom collections import defaultdict, Counter\nfrom multiprocessing import Pool, cpu_count\nimport time\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:21:31.461831Z","iopub.execute_input":"2023-01-22T14:21:31.462537Z","iopub.status.idle":"2023-01-22T14:21:32.674374Z","shell.execute_reply.started":"2023-01-22T14:21:31.462420Z","shell.execute_reply":"2023-01-22T14:21:32.673122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\n\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n        \ndef create_submission(df):\n    df = df.reset_index()\n    df[\"type\"] = df[\"type\"].map(CFG.contentType_mapper)\n    df[\"session_type\"] = df[\"session\"].astype(\"str\") + \"_\" + df[\"type\"].astype(\"str\") + \"s\"\n    df = df[[\"session_type\", \"prediction\"]].rename({\"prediction\": \"labels\"}, axis=1)\n    return df\n\ndef create_get_ts(ts):\n    return int((ts.replace(tzinfo=CFG.tz) - CFG.ts_zero).total_seconds())","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:21:32.676735Z","iopub.execute_input":"2023-01-22T14:21:32.677217Z","iopub.status.idle":"2023-01-22T14:21:32.693131Z","shell.execute_reply.started":"2023-01-22T14:21:32.677167Z","shell.execute_reply":"2023-01-22T14:21:32.691966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    local = False\n    debug = False\n    tz = datetime.timezone.utc\n    ts_zero = datetime.datetime(1970, 1, 1, tzinfo=tz)\n    contentType_mapper = pd.Series([\"clicks\", \"carts\", \"orders\"], index=[0, 1, 2])\n    target_weight = (0.1, 0.3, 0.6)\n\nif CFG.local:\n    CFG.folder_path = \"./dataset/\"\nelse:\n    CFG.folder_path = \"/kaggle/input/\"","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:21:32.694874Z","iopub.execute_input":"2023-01-22T14:21:32.695377Z","iopub.status.idle":"2023-01-22T14:21:32.721152Z","shell.execute_reply.started":"2023-01-22T14:21:32.695326Z","shell.execute_reply":"2023-01-22T14:21:32.719677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"code","source":"fraction_of_sessions_to_use = 0.01 if CFG.debug else 1\n\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\nif fraction_of_sessions_to_use != 1:\n    lucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_test = test[test.session.isin(lucky_sessions_test)]\nelse:\n    subset_of_test = test\n\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])\n\nsubsets = subset_of_test\nsessions = subsets.session.unique()","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:40:15.019863Z","iopub.execute_input":"2023-01-22T14:40:15.021141Z","iopub.status.idle":"2023-01-22T14:40:16.008916Z","shell.execute_reply.started":"2023-01-22T14:40:15.021087Z","shell.execute_reply":"2023-01-22T14:40:16.007587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test 데이터 세션별 aid 및 event 리스트화\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = subsets.reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = subsets.reset_index(drop=True).groupby('session')['type'].apply(list)\ntest_session_time_interval = subsets.reset_index(drop=True).groupby('session')['ts'].apply(lambda x: [np.log1p(1 / (i+1)) for i in ((x.max() - x) / 3600).values])","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:40:16.305598Z","iopub.execute_input":"2023-01-22T14:40:16.306016Z","iopub.status.idle":"2023-01-22T14:51:10.072841Z","shell.execute_reply.started":"2023-01-22T14:40:16.305982Z","shell.execute_reply":"2023-01-22T14:51:10.071508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test, subset_of_test; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:25:04.183751Z","iopub.status.idle":"2023-01-22T14:25:04.184157Z","shell.execute_reply.started":"2023-01-22T14:25:04.183962Z","shell.execute_reply":"2023-01-22T14:25:04.183982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling & Inference","metadata":{}},{"cell_type":"code","source":"n_aids = 20\n\n# CORE HYPER-PARAMERTER 1 - weight by event types\ntype_weight_multipliers = {0: 0.1, 1: 0.6, 2: 0.3}\n\noutput = {\n    \"session\": [],\n    \"type\": [],\n    \"rec\": [],\n    \"score\": [],\n}\n\nfor SESS, AIDs, types, time_inteval in tqdm(zip(test_session_AIDs.index, test_session_AIDs.values, test_session_types.values, test_session_time_interval.values), total=len(test_session_AIDs)):\n\n    candidates = Counter()\n    for aid, w, t in zip(AIDs[::-1], time_inteval[::-1], types[::-1]):\n        candidates.update({aid: w * type_weight_multipliers[t]})\n    rec, score = zip(*candidates.most_common(n_aids))\n    rec_list = \" \".join([str(k) for k in list(rec)])\n    score_list = \" \".join([str(k) for k in list(np.round(score, 5))])                    \n        \n    output[\"session\"].extend([SESS] * 3)\n    output[\"type\"].extend([0, 1, 2])\n    output[\"rec\"].extend([rec_list] * 3)\n    output[\"score\"].extend([score_list] * 3)\n\noutput = pd.DataFrame(output).set_index([\"session\", \"type\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-08T05:00:07.319837Z","iopub.execute_input":"2023-01-08T05:00:07.320372Z","iopub.status.idle":"2023-01-08T05:00:07.342487Z","shell.execute_reply.started":"2023-01-08T05:00:07.320331Z","shell.execute_reply":"2023-01-08T05:00:07.341061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"execution":{"iopub.status.busy":"2023-01-08T04:59:29.773604Z","iopub.execute_input":"2023-01-08T04:59:29.774112Z","iopub.status.idle":"2023-01-08T04:59:29.795237Z","shell.execute_reply.started":"2023-01-08T04:59:29.774074Z","shell.execute_reply":"2023-01-08T04:59:29.793748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.reset_index().to_parquet(\"./raw_output.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-01-02T04:07:55.383370Z","iopub.execute_input":"2023-01-02T04:07:55.383752Z","iopub.status.idle":"2023-01-02T04:07:55.433046Z","shell.execute_reply.started":"2023-01-02T04:07:55.383719Z","shell.execute_reply":"2023-01-02T04:07:55.431759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"output[\"session_type\"] = [str(i[0]) + \"_\" + str(CFG.contentType_mapper[i[1]]) for i in output.index]","metadata":{"execution":{"iopub.status.busy":"2023-01-02T04:07:55.434998Z","iopub.execute_input":"2023-01-02T04:07:55.435673Z","iopub.status.idle":"2023-01-02T04:07:55.891963Z","shell.execute_reply.started":"2023-01-02T04:07:55.435620Z","shell.execute_reply":"2023-01-02T04:07:55.890476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/otto-recommender-system/sample_submission.csv\")\nsubmission = submission.set_index(\"session_type\")\nsubmission.loc[output[\"session_type\"].values, \"labels\"] = output[\"rec\"].values\nsubmission = submission.reset_index()\nsubmission.to_csv(\"./submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T04:07:55.894109Z","iopub.execute_input":"2023-01-02T04:07:55.894658Z","iopub.status.idle":"2023-01-02T04:08:15.639844Z","shell.execute_reply.started":"2023-01-02T04:07:55.894608Z","shell.execute_reply":"2023-01-02T04:08:15.637971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-01-02T04:08:15.641463Z","iopub.execute_input":"2023-01-02T04:08:15.642972Z","iopub.status.idle":"2023-01-02T04:08:15.664430Z","shell.execute_reply.started":"2023-01-02T04:08:15.642888Z","shell.execute_reply":"2023-01-02T04:08:15.662896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot on Length of Labels","metadata":{}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nplt.hist([len(l.split(\" \")) for l in submission[\"labels\"]]);\nplt.suptitle('Distribution of predicted sequence lengths');","metadata":{"execution":{"iopub.status.busy":"2023-01-02T04:08:15.666212Z","iopub.execute_input":"2023-01-02T04:08:15.666632Z","iopub.status.idle":"2023-01-02T04:08:36.224494Z","shell.execute_reply.started":"2023-01-02T04:08:15.666596Z","shell.execute_reply":"2023-01-02T04:08:36.222523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}