{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About\n\nWe have to create negative examples and prediction candidates in this competition.  \nFor the puropose, I pre-extract top-100 similar aids using aid2vec model using Word2Vec.\n\nI use @columbia2131 's [Dataset](https://www.kaggle.com/datasets/columbia2131/otto-chunk-data-inparquet-format) and refer @takaito 's [Word2Vec Tutorial](https://www.kaggle.com/code/takaito/otto-word2vec-tutorial/), great thanks!","metadata":{}},{"cell_type":"markdown","source":"## Preparation","metadata":{}},{"cell_type":"markdown","source":"### load libraries","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport sys\nimport typing as tp\nimport pickle as pkl\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm, trange\n\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:48:52.163992Z","iopub.execute_input":"2022-11-04T20:48:52.164416Z","iopub.status.idle":"2022-11-04T20:48:52.171213Z","shell.execute_reply.started":"2022-11-04T20:48:52.164382Z","shell.execute_reply":"2022-11-04T20:48:52.169934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"otto-recommender-system\"\n\nTRAIN_PARQUET = INPUT / \"otto-chunk-data-inparquet-format\" / \"train_parquet\"\nTEST_PARQUET = INPUT / \"otto-chunk-data-inparquet-format\" / \"test_parquet\"\n\nfor f in DATA.iterdir():\n    print(f.name)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-04T20:24:34.090660Z","iopub.execute_input":"2022-11-04T20:24:34.091835Z","iopub.status.idle":"2022-11-04T20:24:34.099675Z","shell.execute_reply.started":"2022-11-04T20:24:34.091795Z","shell.execute_reply":"2022-11-04T20:24:34.098153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training aid2vec model by Word2Vec","metadata":{}},{"cell_type":"code","source":"def extract_sentences_from_chunk(sentences: tp.List[tp.List[int]], chunk: pd.DataFrame):\n    \"\"\"\"\"\"\n    session_ids = sorted(chunk[\"session\"].unique())\n    session_count = chunk[\"session\"].value_counts()\n    aids = chunk[\"aid\"].to_list()\n    \n    pointor = 0\n    for session_id in session_ids:\n        session_length = session_count[session_id]\n        if session_length > 1:  # ignore single word (i.e. single action) sentences\n            sentences.append(aids[pointor: pointor + session_length])\n        pointor += session_length\n\n\ndef extract_sentences():\n    train_parquet_paths = sorted(TRAIN_PARQUET.glob(\"*parquet\"))\n    test_parquet_paths = sorted(TEST_PARQUET.glob(\"*parquet\"))                   \n    sentences = []\n    \n    for parquet_path in tqdm(train_parquet_paths):\n        chunk = pd.read_parquet(parquet_path)\n        extract_sentences_from_chunk(sentences, chunk)\n        \n    for parquet_path in tqdm(test_parquet_paths):\n        chunk = pd.read_parquet(parquet_path)\n        extract_sentences_from_chunk(sentences, chunk)\n        \n    return sentences\n\n\ndef training_word2vec_model(sentences:tp.List[tp.List[int]], vector_size: int=128):\n    import hashlib\n    from gensim.models import word2vec\n    from gensim.models import KeyedVectors\n    os.environ[\"PYTHONHASHSEED\"] = str(42)\n    def hashfxn(x):\n        return int(hashlib.md5(str(x).encode()).hexdigest(), 16)\n            \n    model = word2vec.Word2Vec(\n        sentences=sentences, vector_size=vector_size, window=10, min_count=1,\n        sg=0, workers=-1, seed=42, hashfxn=hashfxn, sorted_vocab=1)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:28:09.107580Z","iopub.execute_input":"2022-11-04T20:28:09.108008Z","iopub.status.idle":"2022-11-04T20:28:09.122176Z","shell.execute_reply.started":"2022-11-04T20:28:09.107973Z","shell.execute_reply":"2022-11-04T20:28:09.120894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = extract_sentences()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:28:42.791030Z","iopub.execute_input":"2022-11-04T20:28:42.791433Z","iopub.status.idle":"2022-11-04T20:32:01.097408Z","shell.execute_reply.started":"2022-11-04T20:28:42.791402Z","shell.execute_reply":"2022-11-04T20:32:01.096130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\naid2vec_model = training_word2vec_model(sentences, vector_size=64)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:58:57.120955Z","iopub.execute_input":"2022-11-04T20:58:57.121419Z","iopub.status.idle":"2022-11-04T21:02:01.709892Z","shell.execute_reply.started":"2022-11-04T20:58:57.121383Z","shell.execute_reply":"2022-11-04T21:02:01.217331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aid2vec_model.wv.vectors.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T21:02:25.084119Z","iopub.execute_input":"2022-11-04T21:02:25.084592Z","iopub.status.idle":"2022-11-04T21:02:26.617836Z","shell.execute_reply.started":"2022-11-04T21:02:25.084554Z","shell.execute_reply":"2022-11-04T21:02:26.519872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ","metadata":{}},{"cell_type":"markdown","source":"## Pre-extract similar aids","metadata":{}},{"cell_type":"code","source":"similar_aids_dict = {}\nfor aid_i in trange(aid2vec_model.wv.vectors.shape[0]):\n    similar_aids = aid2vec_model.wv.similar_by_word(aid_i, topn=100, restrict_vocab=300000)\n    similar_aids = [aid_j for aid_j, _ in similar_aids]\n    if aid_i not in set(similar_aids):\n        similar_aids = [aid_i] + similar_aids[:-1]  # add itself\n    \n    similar_aids_dict[aid_i] = similar_aids","metadata":{"execution":{"iopub.status.busy":"2022-11-04T21:05:57.059245Z","iopub.execute_input":"2022-11-04T21:05:57.059669Z","iopub.status.idle":"2022-11-04T21:06:54.794838Z","shell.execute_reply.started":"2022-11-04T21:05:57.059632Z","shell.execute_reply":"2022-11-04T21:06:54.789187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save the result","metadata":{}},{"cell_type":"code","source":"aid2vec_model.wv.save_word2vec_format('aid2vec_64dim.bin', binary=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:52:41.440840Z","iopub.execute_input":"2022-11-04T20:52:41.441256Z","iopub.status.idle":"2022-11-04T20:52:52.449889Z","shell.execute_reply.started":"2022-11-04T20:52:41.441221Z","shell.execute_reply":"2022-11-04T20:52:52.448644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"similar_aids_dict.pkl\", \"wb\") as fw:\n    pkl.dump(similar_aids_dict, fw)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:53:20.594929Z","iopub.execute_input":"2022-11-04T20:53:20.595333Z","iopub.status.idle":"2022-11-04T20:53:21.598677Z","shell.execute_reply.started":"2022-11-04T20:53:20.595302Z","shell.execute_reply":"2022-11-04T20:53:21.597424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EOF","metadata":{}}]}