{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n<br><br/>\n**User-Based Filtering**\n\nFor retrieval this kernel use a approach which is view of user, similar to CBF based algorithm (Co-Visitation Matrix).\n\nContrast to co-visitation, this algorithm calculate the counts of top N interacted users' aid history.\n\n*In inference if counted aids from UBF algorithm exist, drop the history aids for evalutate precise recall value*\n<br><br/><br><br/>\n**LB Score**\n\nUBF Standalone : 0.046\n\nUBF applied when AIDs < 20 : 0.520","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ[\"PYTHONIOENCODING\"] = \"utf8\"\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\n\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nimport random as rnd\nimport shutil\nimport gc\nimport datetime\nfrom collections import defaultdict, Counter\nfrom tqdm import tqdm\nfrom multiprocessing import Pool, cpu_count\nimport time\nfrom itertools import chain\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-01-30T09:00:14.859068Z","iopub.execute_input":"2023-01-30T09:00:14.859812Z","iopub.status.idle":"2023-01-30T09:00:16.023692Z","shell.execute_reply.started":"2023-01-30T09:00:14.859709Z","shell.execute_reply":"2023-01-30T09:00:16.022331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\n\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n        \ndef create_submission(df):\n    df = df.reset_index()\n    df[\"type\"] = df[\"type\"].map(CFG.contentType_mapper)\n    df[\"session_type\"] = df[\"session\"].astype(\"str\") + \"_\" + df[\"type\"].astype(\"str\") + \"s\"\n    df = df[[\"session_type\", \"prediction\"]].rename({\"prediction\": \"labels\"}, axis=1)\n    return df\n\ndef create_get_ts(ts):\n    return int((ts.replace(tzinfo=CFG.tz) - CFG.ts_zero).total_seconds())","metadata":{"execution":{"iopub.status.busy":"2023-01-30T09:00:16.026217Z","iopub.execute_input":"2023-01-30T09:00:16.026724Z","iopub.status.idle":"2023-01-30T09:00:16.041166Z","shell.execute_reply.started":"2023-01-30T09:00:16.026677Z","shell.execute_reply":"2023-01-30T09:00:16.040039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    local = False\n    debug = False\n    tz = datetime.timezone.utc\n    ts_zero = datetime.datetime(1970, 1, 1, tzinfo=tz)\n    contentType_mapper = pd.Series([\"clicks\", \"carts\", \"orders\"], index=[0, 1, 2])\n    target_weight = (0.1, 0.3, 0.6)\n\nif CFG.local:\n    CFG.folder_path = \"./dataset/\"\nelse:\n    CFG.folder_path = \"/kaggle/input/\"","metadata":{"execution":{"iopub.status.busy":"2023-01-30T09:00:16.042976Z","iopub.execute_input":"2023-01-30T09:00:16.043743Z","iopub.status.idle":"2023-01-30T09:00:16.065654Z","shell.execute_reply.started":"2023-01-30T09:00:16.043698Z","shell.execute_reply":"2023-01-30T09:00:16.064270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"code","source":"fraction_of_sessions_to_use = 0.1 if CFG.debug else 1\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\nif fraction_of_sessions_to_use != 1:\n    lucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_train = train[train.session.isin(lucky_sessions_train)]\n    \n    lucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_test = test[test.session.isin(lucky_sessions_test)]\nelse:\n    subset_of_train = train\n    subset_of_test = test\n\nsubset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])\n\n# Concat train & test data\n# This is not information leakage because of considering the predctions session by session\nsubsets = pd.concat([subset_of_train, subset_of_test])\nsessions = subsets.session.unique()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T09:00:16.067218Z","iopub.execute_input":"2023-01-30T09:00:16.067728Z","iopub.status.idle":"2023-01-30T09:01:06.833901Z","shell.execute_reply.started":"2023-01-30T09:00:16.067683Z","shell.execute_reply":"2023-01-30T09:01:06.831650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate Best Interacted User by Items","metadata":{}},{"cell_type":"code","source":"chunk_size = 16384 * 2\nn_aids = 20\n\n# CORE: You can tune this hyper-parameter\ntype_weight_multipliers = {0: 0.1, 1: 0.3, 2: 0.6}\nnext_AIDs = defaultdict(Counter)\n\n# Split data by Chunks\nfor i in tqdm(range(0, sessions.shape[0], chunk_size), total=len(range(0, sessions.shape[0], chunk_size))):\n    # Get subset from entire dataset\n    current_chunk = subsets.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n    # Drop duplicated action by user on same product\n    # Get only last interacted 20 aid (removing the complexity)\n    current_chunk = current_chunk.drop_duplicates().groupby('session', as_index=False).nth(list(range(-n_aids, 0))).reset_index(drop=True)\n    # Calculate interacted item's score by type\n    # If a user buy the item, we can consider the user love it.\n    for aid, sess, t in zip(current_chunk[\"aid\"], current_chunk[\"session\"], current_chunk[\"type\"]):\n        next_AIDs[aid][sess] += type_weight_multipliers[t]","metadata":{"execution":{"iopub.status.busy":"2023-01-30T09:01:32.855830Z","iopub.execute_input":"2023-01-30T09:01:32.856308Z","iopub.status.idle":"2023-01-30T09:01:55.684158Z","shell.execute_reply.started":"2023-01-30T09:01:32.856260Z","shell.execute_reply":"2023-01-30T09:01:55.682155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = subset_of_test.reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = subset_of_test.reset_index(drop=True).groupby('session')['type'].apply(list)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:02:13.327067Z","iopub.execute_input":"2023-01-26T07:02:13.327554Z","iopub.status.idle":"2023-01-26T07:04:34.258360Z","shell.execute_reply.started":"2023-01-26T07:02:13.327511Z","shell.execute_reply":"2023-01-26T07:04:34.257135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, subset_of_train; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:04:34.259908Z","iopub.execute_input":"2023-01-26T07:04:34.260301Z","iopub.status.idle":"2023-01-26T07:04:49.891674Z","shell.execute_reply.started":"2023-01-26T07:04:34.260264Z","shell.execute_reply":"2023-01-26T07:04:49.890447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get user's aid history data for inference\nuser_aids = subsets.reset_index(drop=True).groupby('session')[\"aid\"].nth(list(range(-n_aids, 0)))","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:04:49.893250Z","iopub.execute_input":"2023-01-26T07:04:49.894250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling & Inference","metadata":{}},{"cell_type":"code","source":"n_aids = 20\n\noutput = {\n    \"session\": [],\n    \"type\": [],\n    \"rec\": [],\n    \"score\": [],\n}\n\nfor SESS, AIDs, types in tqdm(zip(test_session_AIDs.index, test_session_AIDs.values, test_session_types.values), total=len(test_session_AIDs.index)):\n    \n    candidates = Counter()\n    for aid in AIDs[::-1]:\n        # Get top N most interacted user (N=4)\n        # Median of user interaction is 5. So I decide N=4 (4 * 5 is about 20)\n        # CORE: You can \n        for i, j in next_AIDs[aid].most_common(4):\n            # The user's interacted record is absoultly in UBF dictionary.\n            # So skip inference on me. (If not, this is just same as inferening with the user history items !)\n            if (i == SESS) or (j == 1):\n                continue\n            tmp_aids = user_aids.loc[[i]].to_list()\n            tmp_aids = Counter(tmp_aids)\n            candidates += Counter(tmp_aids)\n    \n    # Do re-drop process because other people also co-interacted with the user's aid history\n    # If not, this is just same as inferening with the user history items\n    for aid in AIDs[::-1]:\n        if aid in candidates: del candidates[aid]\n    \n    # But if retrieved aids is zero, infer with the user history items.\n    # This process is a little wierd but, I wanted to get recall without the user history aid\n    if len(candidates) == 0: candidates = Counter(AIDs)\n    \n    rec, score = zip(*candidates.most_common(n_aids))\n    \n    output[\"session\"].extend([SESS] * 3)\n    output[\"type\"].extend([0, 1, 2])\n    output[\"rec\"].extend([\" \".join(pd.Series(rec, dtype=\"str\").values)] * 3)\n    output[\"score\"].extend([\" \".join(pd.Series(score, dtype=\"str\").values)] * 3)\n\noutput = pd.DataFrame(output).set_index([\"session\", \"type\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:37:29.385661Z","iopub.execute_input":"2023-01-20T10:37:29.386180Z","iopub.status.idle":"2023-01-20T10:37:29.414828Z","shell.execute_reply.started":"2023-01-20T10:37:29.386136Z","shell.execute_reply":"2023-01-20T10:37:29.413416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"execution":{"iopub.status.busy":"2023-01-19T08:12:26.360147Z","iopub.status.idle":"2023-01-19T08:12:26.360586Z","shell.execute_reply.started":"2023-01-19T08:12:26.360386Z","shell.execute_reply":"2023-01-19T08:12:26.360405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.reset_index().to_parquet(\"./raw_output.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-01-19T08:12:26.362354Z","iopub.status.idle":"2023-01-19T08:12:26.362815Z","shell.execute_reply.started":"2023-01-19T08:12:26.362594Z","shell.execute_reply":"2023-01-19T08:12:26.362627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"output[\"session_type\"] = [str(i[0]) + \"_\" + str(CFG.contentType_mapper[i[1]]) for i in output.index]","metadata":{"execution":{"iopub.status.busy":"2023-01-19T08:12:26.364866Z","iopub.status.idle":"2023-01-19T08:12:26.365319Z","shell.execute_reply.started":"2023-01-19T08:12:26.365119Z","shell.execute_reply":"2023-01-19T08:12:26.365139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/otto-recommender-system/sample_submission.csv\")\nsubmission = submission.set_index(\"session_type\")\nsubmission.loc[output[\"session_type\"].values, \"labels\"] = output[\"rec\"].values\nsubmission = submission.reset_index()\nsubmission.to_csv(\"./submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-19T08:12:26.366940Z","iopub.status.idle":"2023-01-19T08:12:26.367357Z","shell.execute_reply.started":"2023-01-19T08:12:26.367158Z","shell.execute_reply":"2023-01-19T08:12:26.367177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-01-19T08:12:26.368601Z","iopub.status.idle":"2023-01-19T08:12:26.369042Z","shell.execute_reply.started":"2023-01-19T08:12:26.368847Z","shell.execute_reply":"2023-01-19T08:12:26.368866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}