{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install faiss-cpu --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:08.356748Z","iopub.execute_input":"2026-07-09T23:57:08.357472Z","iopub.status.idle":"2026-07-09T23:57:13.140700Z","shell.execute_reply.started":"2026-07-09T23:57:08.357431Z","shell.execute_reply":"2026-07-09T23:57:13.139404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-13T06:20:21.905478Z","iopub.execute_input":"2026-07-13T06:20:21.906629Z","iopub.status.idle":"2026-07-13T06:20:23.546899Z","shell.execute_reply.started":"2026-07-13T06:20:21.906586Z","shell.execute_reply":"2026-07-13T06:20:23.545853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nfrom datetime import datetime\n\nTRAIN_PATH = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\n\ntotal_sessions = sum(1 for _ in open(TRAIN_PATH))\nprint(f\"Всего сессий: {total_sessions}\")\n\ntarget_count = 4000000\nstart_line = max(0, total_sessions - target_count)\n\nprint(f\"Анализ последних {target_count} сессий (начиная со строки {start_line})\")\n\nall_min_ts = float('inf')\nall_max_ts = float('-inf')\nlast_min_ts = float('inf')\nlast_max_ts = float('-inf')\n\nwith open(TRAIN_PATH) as f:\n    for i, line in enumerate(f):\n        record = json.loads(line)\n        \n        session_ts = [event['ts'] for event in record['events']]\n        if session_ts:\n            session_min = min(session_ts)\n            session_max = max(session_ts)\n            \n            if session_min < all_min_ts:\n                all_min_ts = session_min\n            if session_max > all_max_ts:\n                all_max_ts = session_max\n            \n            if i >= start_line:\n                if session_min < last_min_ts:\n                    last_min_ts = session_min\n                if session_max > last_max_ts:\n                    last_max_ts = session_max\n        \n        if (i + 1) % 1000000 == 0:\n            print(f\"  Обработано {i + 1} сессий\")\n\nprint(f\"\\nВсего сессий: {total_sessions}\")\nprint(f\"\\nВесь train.jsonl:\")\nprint(f\"  Мин: {datetime.fromtimestamp(all_min_ts / 1000)}\")\nprint(f\"  Макс: {datetime.fromtimestamp(all_max_ts / 1000)}\")\nprint(f\"  Длительность: {(all_max_ts - all_min_ts) / 86400000:.1f} дней\")\n\nprint(f\"\\nПоследние {target_count} сессий:\")\nprint(f\"  Мин: {datetime.fromtimestamp(last_min_ts / 1000)}\")\nprint(f\"  Макс: {datetime.fromtimestamp(last_max_ts / 1000)}\")\nprint(f\"  Длительность: {(last_max_ts - last_min_ts) / 86400000:.1f} дней\")\nprint(f\"  Покрытие: {(last_max_ts - last_min_ts) / (all_max_ts - all_min_ts) * 100:.1f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:13.749071Z","iopub.execute_input":"2026-07-09T23:57:13.749346Z","iopub.status.idle":"2026-07-09T23:57:17.339917Z","shell.execute_reply.started":"2026-07-09T23:57:13.749319Z","shell.execute_reply":"2026-07-09T23:57:17.338113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nfrom collections import defaultdict\n\nTRAIN_PATH = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\n\ncounts = defaultdict(int)\ntotal_events = 0\n\nwith open(TRAIN_PATH) as f:\n    for line in f:\n        record = json.loads(line)\n        \n        for event in record['events']:\n            counts[event['type']] += 1\n            total_events += 1\n\nprint(\"Статистика по типам событий\")\nprint(f\"Всего событий: {total_events}\")\nprint(f\"\\nПо типам:\")\nfor event_type in ['clicks', 'carts', 'orders']:\n    count = counts[event_type]\n    pct = count / total_events * 100\n    avg_per_session = count / total_sessions\n    print(f\"  {event_type:8s}: {count:>12,} ({pct:5.2f}%) | среднее {avg_per_session:.2f} на сессию\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:17.340583Z","iopub.status.idle":"2026-07-09T23:57:17.340909Z","shell.execute_reply.started":"2026-07-09T23:57:17.340758Z","shell.execute_reply":"2026-07-09T23:57:17.340779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\nunique_aids = set()\n\nwith open(TRAIN_PATH) as f:\n    for i, line in enumerate(f):\n        if i % 1000000 == 0:\n            print(f\"Обработано {i} сессий\")\n        record = json.loads(line)\n        for event in record['events']:\n            unique_aids.add(str(event['aid']))\n\nprint(f\"\\nУникальных товаров: {len(unique_aids)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:17.342300Z","iopub.status.idle":"2026-07-09T23:57:17.342710Z","shell.execute_reply.started":"2026-07-09T23:57:17.342504Z","shell.execute_reply":"2026-07-09T23:57:17.342530Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Генерация кандидатов","metadata":{}},{"cell_type":"markdown","source":"#### Кандидаты берутся из: Item2Vec, добавляются из итемов сессии и если сгенерировано недостаточно дополняются популярными ","metadata":{}},{"cell_type":"markdown","source":"#### Я использовал Item2Vec(3 Item2Vec под каждый тип события) вместо более сложных нейросетей и других методов кандидатогенерации из-за ограничений оперативной памяти и достаточно быстрой скорости обучения. Какие альтернативы рассматривал: 1)Я пробовал реализовать co-visitation матрицы, но все время упирался в оперативку.  2)Обучить SasRec на сэмплированых данных (3 Sasrec под каждый тип события). Отказался из-за того что в сессиях в среднем 1.31 добавление в корзину и 0.4 заказа, т.е SasRec для этого избыточен. Item2Vec позволил мне обучить модель на полном 90% train dataset (11.6M сессий, 216M событий) за 4,5 часа на CPU, используя 15 GB RAM. Для быстрого инференса я построил FAISS HNSW индекс.","metadata":{}},{"cell_type":"markdown","source":"## Обучение Item2Vec","metadata":{}},{"cell_type":"code","source":"total_sessions=12899779\n\nval_size = int(total_sessions * 0.1)\ntrain_size = total_sessions - val_size","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T12:59:18.591388Z","iopub.execute_input":"2026-07-14T12:59:18.591775Z","iopub.status.idle":"2026-07-14T12:59:18.628675Z","shell.execute_reply.started":"2026-07-14T12:59:18.591744Z","shell.execute_reply":"2026-07-14T12:59:18.627668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"import json\nfrom collections import defaultdict\nfrom gensim.models import Word2Vec\nimport os\n\nPATH = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\nOUT_DIR = '/kaggle/working'\nos.makedirs(OUT_DIR, exist_ok=True)\n\nfiles = {t: open(f\"{OUT_DIR}/sessions_{t}.txt\", \"w\") for t in ['clicks', 'carts', 'orders']}\nwith open(PATH) as f:\n    for i, line in enumerate(f):\n        if i >= train_size:\n            break\n            \n        if i % 100000 == 0:\n            print(f'обработано {i} сессий')\n        events = json.loads(line)['events']\n        by_type = defaultdict(list)\n        for e in events:\n            by_type[e['type']].append(str(e['aid']))\n        for t in files:\n            if by_type[t]:\n                files[t].write(' '.join(by_type[t]) + '\\n')\nfor f in files.values(): f.close()","metadata":{"execution":{"iopub.status.busy":"2026-07-09T23:57:17.345221Z","iopub.status.idle":"2026-07-09T23:57:17.345645Z","shell.execute_reply.started":"2026-07-09T23:57:17.345433Z","shell.execute_reply":"2026-07-09T23:57:17.345459Z"}}},{"cell_type":"markdown","source":"class SessionReader:\n    def __init__(self, filepath):\n        self.filepath = filepath\n    def __iter__(self):\n        with open(self.filepath, 'r') as f:\n            for line in f:\n                yield line.split()\n\n# workers стоит 1 потому что запускал на kaggle cpu\nconfigs = {\n    'clicks': {'vector_size': 64, 'window': 10, 'min_count': 5, 'epochs': 2, 'workers': 1},\n    'carts':  {'vector_size': 64, 'window': 8,  'min_count': 3, 'epochs': 4, 'workers': 1},\n    'orders': {'vector_size': 64, 'window': 6,  'min_count': 2, 'epochs': 4, 'workers': 1}\n}\n\nfor t, cfg in configs.items():\n    print(f\"Training {t}\")\n    model = Word2Vec(\n        SessionReader(f\"{OUT_DIR}/sessions_{t}.txt\"), \n        sg=1, negative=5, sample=1e-5, seed=11, **cfg\n    )\n    model.save(f\"{OUT_DIR}/item2vec_{t}.model\")\n    print(f\"Saved {t}\")","metadata":{"execution":{"iopub.status.busy":"2026-07-09T23:57:17.346978Z","iopub.status.idle":"2026-07-09T23:57:17.347386Z","shell.execute_reply.started":"2026-07-09T23:57:17.347171Z","shell.execute_reply":"2026-07-09T23:57:17.347196Z"}}},{"cell_type":"markdown","source":"#### To Do: добавить валидацию","metadata":{}},{"cell_type":"markdown","source":"### Ниже предсказания Item2Vec на тестовых данных. private score=0.03027, public score=0.03006","metadata":{}},{"cell_type":"code","source":"!pip install faiss-cpu --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:17.348569Z","iopub.status.idle":"2026-07-09T23:57:17.348902Z","shell.execute_reply.started":"2026-07-09T23:57:17.348769Z","shell.execute_reply":"2026-07-09T23:57:17.348786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport faiss\nfrom gensim.models import Word2Vec\n\nMODELS_DIR = '/kaggle/input/datasets/igorvorobev7892/item2vec-models-for-recproj2'\n\ndef load_model_and_build_hnsw_index(model_path, M=16, efConstruction=200):\n    model = Word2Vec.load(model_path)\n    vectors = model.wv.vectors.astype('float32')\n    faiss.normalize_L2(vectors)\n    \n    dim = vectors.shape[1]\n    \n    index = faiss.IndexHNSWFlat(dim, M, faiss.METRIC_INNER_PRODUCT)\n    index.hnsw.efConstruction = efConstruction\n    index.add(vectors)\n    index.hnsw.efSearch = 128\n    \n    idx_to_aid = model.wv.index_to_key\n    return model, index, idx_to_aid\n\nmodels = {}\nindexes = {}\nidx_to_aids = {}\n\nfor t in ['clicks', 'carts', 'orders']:\n    model, index, idx_to_aid = load_model_and_build_hnsw_index(\n        f'{MODELS_DIR}/item2vec_{t}.model'\n    )\n    models[t] = model\n    indexes[t] = index\n    idx_to_aids[t] = idx_to_aid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:17.350036Z","iopub.status.idle":"2026-07-09T23:57:17.350478Z","shell.execute_reply.started":"2026-07-09T23:57:17.350229Z","shell.execute_reply":"2026-07-09T23:57:17.350254Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"def predict_batch(model, faiss_index, idx_to_aid, all_sessions_aids, all_session_timestamps, top_k=20, last_n=10):\n    session_vectors = []\n    \n    for session_aids, session_timestamps in zip(all_sessions_aids, all_session_timestamps):\n        valid_pairs = []\n        for aid, ts in zip(session_aids, session_timestamps):\n            if aid in model.wv.key_to_index:\n                valid_pairs.append((aid, ts))\n        \n        if not valid_pairs:\n            session_vectors.append(np.zeros(model.vector_size, dtype='float32'))\n            continue\n        \n        recent_pairs = valid_pairs[-last_n:] if len(valid_pairs) > last_n else valid_pairs\n        \n        vectors = []\n        weights = []\n        \n        min_ts = min(ts for _, ts in recent_pairs)\n        max_ts = max(ts for _, ts in recent_pairs)\n        time_range = max_ts - min_ts\n        \n        for aid, ts in recent_pairs:\n            vectors.append(model.wv[aid])\n            \n            if time_range > 0:\n                weight = (ts - min_ts) / time_range\n            else:\n                weight = 1.0\n            weights.append(weight)\n        \n        weights = np.array(weights, dtype=np.float32)\n        weights = weights / weights.sum()\n        \n        session_vec = np.average(vectors, axis=0, weights=weights).astype('float32')\n        session_vectors.append(session_vec)\n    \n    query_matrix = np.array(session_vectors, dtype='float32')\n    faiss.normalize_L2(query_matrix)\n    \n    D, I = faiss_index.search(query_matrix, top_k * 3)\n    \n    results = []\n    for session_aids, indices in zip(all_sessions_aids, I):\n        session_set = set(session_aids)\n        recs = []\n        for idx in indices:\n            if idx < 0:\n                continue\n            aid = idx_to_aid[idx]\n            if aid not in session_set:\n                recs.append(aid)\n                if len(recs) >= top_k:\n                    break\n        results.append(recs)\n    \n    return results","metadata":{"execution":{"iopub.status.busy":"2026-07-09T23:57:17.352214Z","iopub.status.idle":"2026-07-09T23:57:17.352583Z","shell.execute_reply.started":"2026-07-09T23:57:17.352434Z","shell.execute_reply":"2026-07-09T23:57:17.352456Z"}}},{"cell_type":"markdown","source":"TEST_PATH = '/kaggle/input/competitions/otto-recommender-system/test.jsonl'\nSUBMISSION_PATH = '/kaggle/working/submission.csv'\n\ntest_sessions = []\nwith open(TEST_PATH) as f:\n    for line in f:\n        record = json.loads(line)\n        aids = [str(e['aid']) for e in record['events']]\n        timestamps = [e['ts'] for e in record['events']]\n        test_sessions.append((record['session'], aids, timestamps))\n\n\nBATCH_SIZE = 10000\nall_predictions = []\n\nfor event_type in ['clicks', 'carts', 'orders']:\n    print(f\"\\nProcessing {event_type}\")\n    \n    for batch_start in range(0, len(test_sessions), BATCH_SIZE):\n        batch_end = min(batch_start + BATCH_SIZE, len(test_sessions))\n        batch = test_sessions[batch_start:batch_end]\n        \n        session_ids = [s[0] for s in batch]\n        session_aids_list = [s[1] for s in batch]\n        session_timestamps_list = [s[2] for s in batch]\n        \n        batch_results = predict_batch(\n            models[event_type],\n            indexes[event_type],\n            idx_to_aids[event_type],\n            session_aids_list,\n            session_timestamps_list,\n            top_k=20,\n            last_n=10\n        )\n        \n        for session_id, recs in zip(session_ids, batch_results):\n            all_predictions.append(f\"{session_id}_{event_type},{' '.join(recs)}\")\n        \n        if (batch_end // BATCH_SIZE) % 20 == 0:\n            print(f\"  {event_type}: {batch_end}/{len(test_sessions)} sessions\")\n\nwith open(SUBMISSION_PATH, 'w') as f:\n    f.write(\"session_type,labels\\n\")\n    f.write('\\n'.join(all_predictions))","metadata":{"execution":{"iopub.status.busy":"2026-07-09T23:57:17.353809Z","iopub.status.idle":"2026-07-09T23:57:17.354074Z","shell.execute_reply.started":"2026-07-09T23:57:17.353949Z","shell.execute_reply":"2026-07-09T23:57:17.353965Z"}}},{"cell_type":"markdown","source":"## Загрузка Item2Vec из датасета https://www.kaggle.com/datasets/igorvorobev7892/item2vec-models-for-recproj2 (обученные выше модели сохранил туда)","metadata":{}},{"cell_type":"code","source":"!pip install faiss-cpu --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T12:42:38.538621Z","iopub.execute_input":"2026-07-14T12:42:38.538864Z","iopub.status.idle":"2026-07-14T12:42:45.484719Z","shell.execute_reply.started":"2026-07-14T12:42:38.538839Z","shell.execute_reply":"2026-07-14T12:42:45.483584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport faiss\nfrom gensim.models import Word2Vec\n\nMODELS_DIR = '/kaggle/input/datasets/igorvorobev7892/item2vec-models-for-recproj2'\n\ndef load_model_and_build_hnsw_index(model_path, M=16, efConstruction=200):\n    model = Word2Vec.load(model_path)\n    vectors = model.wv.vectors.astype('float32')\n    faiss.normalize_L2(vectors)\n    \n    dim = vectors.shape[1]\n    \n    index = faiss.IndexHNSWFlat(dim, M, faiss.METRIC_INNER_PRODUCT)\n    index.hnsw.efConstruction = efConstruction\n    index.add(vectors)\n    index.hnsw.efSearch = 128\n    \n    idx_to_aid = model.wv.index_to_key\n    return model, index, idx_to_aid\n\nitem2vec_models = {}\nindexes = {}\nidx_to_aids = {}\n\nfor t in ['clicks', 'carts', 'orders']:\n    model, index, idx_to_aid = load_model_and_build_hnsw_index(\n        f'{MODELS_DIR}/item2vec_{t}.model'\n    )\n    item2vec_models[t] = model\n    indexes[t] = index\n    idx_to_aids[t] = idx_to_aid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T12:42:57.374746Z","iopub.execute_input":"2026-07-14T12:42:57.375090Z","iopub.status.idle":"2026-07-14T12:54:11.885478Z","shell.execute_reply.started":"2026-07-14T12:42:57.375049Z","shell.execute_reply":"2026-07-14T12:54:11.884817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Подсчет популярных итемов","metadata":{}},{"cell_type":"code","source":"import json\nimport pickle\nfrom collections import defaultdict\n\nTRAIN_PATH = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\nOUTPUT_PATH = '/kaggle/working/popular_items.pkl'\n\ntype_weights = {'clicks': 1.0, 'carts': 3.0, 'orders': 6.0}\n\npopularity_by_type = {\n    'clicks': defaultdict(float),\n    'carts': defaultdict(float),\n    'orders': defaultdict(float),\n    'all': defaultdict(float)\n}\n\n\nwith open(TRAIN_PATH) as f:\n    for i, line in enumerate(f):\n        if i >= train_size:\n            break\n        \n        if i % 500000 == 0:\n            print(f\"Обработано: {i}/{train_size} сессий\")\n        \n        record = json.loads(line)\n        \n        for event in record['events']:\n            aid = str(event['aid'])\n            event_type = event['type']\n            weight = type_weights[event_type]\n            \n            popularity_by_type[event_type][aid] += weight\n            popularity_by_type['all'][aid] += weight\n\nTOP_N = 100\npopular_items = {}\n\nfor event_type in ['clicks', 'carts', 'orders', 'all']:\n    sorted_items = sorted(popularity_by_type[event_type].items(), \n                         key=lambda x: x[1], reverse=True)[:TOP_N]\n    popular_items[event_type] = [aid for aid, _ in sorted_items]\n\nprint(f\"\\nСохранение в {OUTPUT_PATH}\")\nwith open(OUTPUT_PATH, 'wb') as f:\n    pickle.dump(popular_items, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T12:59:22.558839Z","iopub.execute_input":"2026-07-14T12:59:22.559654Z","iopub.status.idle":"2026-07-14T13:08:46.169243Z","shell.execute_reply.started":"2026-07-14T12:59:22.559620Z","shell.execute_reply":"2026-07-14T13:08:46.167699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Итоговая функция генерации кандидатов","metadata":{}},{"cell_type":"code","source":"POPULAR_PATH = '/kaggle/working/popular_items.pkl'\n\nwith open(POPULAR_PATH, 'rb') as f:\n    popular_items = pickle.load(f)\n\ndef generate_candidates(session_aids, session_types, session_ts, event_type, top_k=100, last_n=10):\n    scores = defaultdict(float)\n    type_weights = {'clicks': 1.0, 'carts': 3.0, 'orders': 6.0}\n    \n    min_ts = min(session_ts)\n    max_ts = max(session_ts)\n    time_span = max_ts - min_ts\n    \n    aid_scores = defaultdict(float)\n    for aid, t_type, ts in zip(session_aids, session_types, session_ts):\n        w_type = type_weights.get(t_type, 1.0)\n        \n        w_time = (ts - min_ts) / time_span if time_span > 0 else 1.0\n        \n        w = w_type * w_time\n        \n        aid_scores[aid] += w\n    \n    for aid, score in aid_scores.items():\n        scores[aid] += score\n    \n    model = item2vec_models[event_type]\n    index = indexes[event_type]\n    idx_to_aid = idx_to_aids[event_type]\n    \n    if len(session_aids) > last_n:\n        recent_aids = session_aids[-last_n:]\n        recent_types = session_types[-last_n:]\n        recent_ts = session_ts[-last_n:]\n    else:\n        recent_aids = session_aids\n        recent_types = session_types\n        recent_ts = session_ts\n    \n    vectors = []\n    weights = []\n    for aid, t_type, ts in zip(recent_aids, recent_types, recent_ts):\n        if aid in model.wv.key_to_index:\n            w_type = type_weights.get(t_type, 1.0)\n            \n            w_time = (ts - min_ts) / time_span if time_span > 0 else 1.0\n            \n            w = w_type * w_time\n            \n            vectors.append(model.wv[aid])\n            weights.append(w)\n    \n    if vectors:\n        weights = np.array(weights, dtype=np.float32)\n        weight_sum = weights.sum()\n        if weight_sum > 0:\n            weights /= weight_sum\n        else:\n            weights[:] = 1.0 / len(weights)\n        \n        session_vec = np.average(vectors, axis=0, weights=weights).astype('float32').reshape(1, -1)\n        faiss.normalize_L2(session_vec)\n        \n        # Поиск в HNSW индексе (берем с запасом, т.к. часть может быть в сессии)\n        D, I = index.search(session_vec, 2*top_k)\n        \n        for idx, dist in zip(I[0], D[0]):\n            if idx != -1:  # HNSW может вернуть -1, если не нашел соседей\n                aid = idx_to_aid[idx]\n                # Добавляем с весом, зависящим от расстояния\n                # Чем ближе товар, тем выше вес\n                faiss_weight = 1.0 / (1.0 + dist) if dist >= 0 else 1.0\n                scores[aid] += faiss_weight\n    \n    if len(scores) < top_k:\n        for aid in popular_items.get(event_type, [])[:50]:\n            if aid not in scores:\n                scores[aid] += 0.5\n    \n    sorted_candidates = sorted(scores.items(), key=lambda x: x[1], reverse=True)\n    return [aid for aid, _ in sorted_candidates[:top_k]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T13:33:24.162541Z","iopub.execute_input":"2026-07-14T13:33:24.163575Z","iopub.status.idle":"2026-07-14T13:33:24.180651Z","shell.execute_reply.started":"2026-07-14T13:33:24.163534Z","shell.execute_reply":"2026-07-14T13:33:24.179660Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ranking","metadata":{}},{"cell_type":"markdown","source":"## Я выбрал CatBoost, потому что это стандарт для этапа ранжирования. Также моя задача - мультиобъективная оптимизация с жестким дисбалансом (clicks, carts, orders). CatBoost позволяет напрямую использовать sample_weight в функции потерь, чтобы штрафовать модель пропорционально важности событиях.","metadata":{}},{"cell_type":"code","source":"def compute_features_for_session(session_aids, session_types, session_ts, candidates, event_type, \n                                 model, index, idx_to_aid, popular_items):\n    if not candidates:\n        return []\n    \n    session_set = set(session_aids)\n    session_len = len(session_aids)\n    unique_items = len(session_set)\n    \n    first_pos = {}\n    last_pos = {}\n    count_in_session = {}\n    for idx, aid in enumerate(session_aids):\n        count_in_session[aid] = count_in_session.get(aid, 0) + 1\n        if aid not in first_pos:\n            first_pos[aid] = idx\n        last_pos[aid] = idx\n    \n    aid_types = defaultdict(set)\n    for aid, t in zip(session_aids, session_types):\n        aid_types[aid].add(t)\n    \n    type_weights_map = {'clicks': 1.0, 'carts': 3.0, 'orders': 6.0}\n    \n    min_ts = min(session_ts)\n    max_ts = max(session_ts)\n    time_span = max_ts - min_ts\n    \n    vectors = []\n    final_weights = []\n    \n    for aid, t_type, ts in zip(session_aids, session_types, session_ts):\n        if aid in model.wv.key_to_index:\n            w_time = (ts - min_ts) / time_span if time_span > 0 else 1.0\n            w_type = type_weights_map.get(t_type, 1.0)\n            w = w_time * w_type\n            vectors.append(model.wv[aid])\n            final_weights.append(w)\n    \n    session_vec = None\n    if vectors:\n        weights_arr = np.array(final_weights, dtype=np.float32)\n        weight_sum = weights_arr.sum()\n        if weight_sum > 0:\n            weights_arr /= weight_sum\n        else:\n            weights_arr[:] = 1.0 / len(weights_arr)\n        \n        vectors_arr = np.array(vectors, dtype=np.float32)\n        session_vec = np.average(vectors_arr, axis=0, weights=weights_arr)\n        norm = np.linalg.norm(session_vec)\n        if norm > 0:\n            session_vec /= norm\n    \n    valid_candidates = []\n    candidate_vectors = []\n    \n    for aid in candidates:\n        if aid in model.wv.key_to_index:\n            valid_candidates.append(aid)\n            candidate_vectors.append(model.wv[aid])\n    \n    i2v_session_sims = {}\n    i2v_last_sims = {}\n    \n    if candidate_vectors:\n        candidate_matrix = np.array(candidate_vectors, dtype='float32')\n        norms = np.linalg.norm(candidate_matrix, axis=1, keepdims=True)\n        norms[norms == 0] = 1.0\n        candidate_matrix_norm = candidate_matrix / norms\n        \n        # Similarity с эмбеддингом сессии\n        if session_vec is not None:\n            sims = candidate_matrix_norm @ session_vec\n            for aid, sim in zip(valid_candidates, sims):\n                i2v_session_sims[aid] = float(sim)\n        \n        # Similarity с последним товаром сессии\n        last_aid = session_aids[-1]\n        if last_aid in model.wv.key_to_index:\n            last_vec = model.wv[last_aid]\n            last_norm = np.linalg.norm(last_vec)\n            if last_norm > 0:\n                last_vec_norm = last_vec / last_norm\n                sims_last = candidate_matrix_norm @ last_vec_norm\n                for aid, sim in zip(valid_candidates, sims_last):\n                    i2v_last_sims[aid] = float(sim)\n    \n    pop_list = popular_items.get(event_type, [])\n    pop_rank_map = {aid: rank for rank, aid in enumerate(pop_list)}\n    \n    rows = []\n    for aid in candidates:\n        in_session = 1 if aid in session_set else 0\n        pop_rank = pop_rank_map.get(aid, 999)\n        aid_type_set = aid_types.get(aid, set())\n        \n        feat = {\n            'in_session': in_session,\n            'count_in_session': count_in_session.get(aid, 0),\n            'first_pos': first_pos.get(aid, -1),\n            'last_pos': last_pos.get(aid, -1),\n            'recency': last_pos[aid] / max(session_len - 1, 1) if aid in last_pos else -1.0,\n            'was_clicked': 1 if 'clicks' in aid_type_set else 0,\n            'was_carted': 1 if 'carts' in aid_type_set else 0,\n            'was_ordered': 1 if 'orders' in aid_type_set else 0,\n            'type_count': len(aid_type_set),\n            'pop_rank': pop_rank,\n            'pop_score': 1.0 - pop_rank / max(len(pop_list), 1),\n            'in_top_10': 1 if pop_rank < 10 else 0,\n            'in_top_50': 1 if pop_rank < 50 else 0,\n            'session_len': session_len,\n            'session_unique': unique_items,\n            'repeat_ratio': 1.0 - unique_items / max(session_len, 1),\n            'pos_normalized': last_pos[aid] / max(session_len - 1, 1) if in_session else -1.0,\n            'is_last_item': 1 if in_session and last_pos[aid] == session_len - 1 else 0,\n            'is_first_item': 1 if in_session and first_pos[aid] == 0 else 0,\n            'distance_from_end': session_len - 1 - last_pos[aid] if in_session else -1,\n            'i2v_session_sim': i2v_session_sims.get(aid, 0.0),\n            'i2v_last_sim': i2v_last_sims.get(aid, 0.0),\n            'event_type': event_type,\n        }\n        rows.append(feat)\n    \n    return rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T13:10:40.037333Z","iopub.execute_input":"2026-07-14T13:10:40.037691Z","iopub.status.idle":"2026-07-14T13:10:40.059070Z","shell.execute_reply.started":"2026-07-14T13:10:40.037660Z","shell.execute_reply":"2026-07-14T13:10:40.058220Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Обучение на сессиях в которых после разбиения сессии(80% для Input, 20% для позитивов) среди позитивов есть хотя бы 1 добавление в корзину и хотя бы 1 заказ","metadata":{}},{"cell_type":"markdown","source":"## Генерация данных для обучения\n#### негативы берем из кандидатогенерации","metadata":{}},{"cell_type":"code","source":"import random\nimport ujson\nimport gc\nimport pandas as pd\n\nMAX_NEGATIVES_PER_POSITIVE = 5\nMIN_NEGATIVES = 10\nMAX_CANDIDATES_PER_SESSION = 100\n\nBATCH_SIZE = 50000\n\ntrain_chunks = []\nval_chunks = []\n\nfiltered_sessions = 0\nprev_batch_first_ixd = 0\nbatch_num = 0\n\n\ndef save_batch(batch_num, train_chunks, val_chunks):\n    if train_chunks:\n        df_train = pd.concat(train_chunks, ignore_index=True)\n        df_train.to_parquet(f'/kaggle/working/train_data_batch_{batch_num}.parquet', index=False)\n        del df_train\n    \n    if val_chunks:\n        df_val = pd.concat(val_chunks, ignore_index=True)\n        df_val.to_parquet(f'/kaggle/working/val_data_batch_{batch_num}.parquet', index=False)\n        del df_val\n    \n    train_chunks.clear()\n    val_chunks.clear()\n    gc.collect()\n\n\nwith open(TRAIN_PATH) as f:\n    for i, line in enumerate(f):\n        if i % 500000 == 0:\n            print(f\"Прошедших фильтр: {filtered_sessions}. Обработано {i}\")\n        \n        if filtered_sessions != prev_batch_first_ixd and filtered_sessions % BATCH_SIZE == 0:\n            save_batch(batch_num, train_chunks, val_chunks)\n            batch_num += 1\n            prev_batch_first_ixd = filtered_sessions\n        \n        record = ujson.loads(line)\n        events = record['events']\n        \n        split_idx = max(1, int(len(events) * 0.8))\n        input_events = events[:split_idx]\n        pos_events = events[split_idx:]\n\n        pos_by_type = defaultdict(set)\n        for e in pos_events:\n            pos_by_type[e['type']].add(str(e['aid']))\n        \n        has_cart_and_order = bool(pos_by_type['carts'] and pos_by_type['orders'])\n        if not has_cart_and_order:\n            continue\n        \n        input_aids = [str(e['aid']) for e in input_events]\n        input_types = [e['type'] for e in input_events]\n        input_ts = [e['ts'] for e in input_events]\n        \n        filtered_sessions += 1\n        \n        for event_type in ['clicks', 'carts', 'orders']:\n            pos_set = pos_by_type[event_type]\n            if not pos_set:\n                continue\n            \n            num_positives = len(pos_set)\n            \n            candidates = generate_candidates(\n                input_aids, input_types, input_ts, event_type, \n                top_k=80, last_n=10\n            )\n            \n            for pos_aid in pos_set:\n                if pos_aid not in candidates:\n                    candidates.append(pos_aid)\n            \n            positives = [c for c in candidates if c in pos_set]\n            negatives = [c for c in candidates if c not in pos_set]\n            \n            target_negatives = max(\n                num_positives * MAX_NEGATIVES_PER_POSITIVE,\n                MIN_NEGATIVES\n            )\n            \n            if len(negatives) > target_negatives:\n                negatives = random.sample(negatives, target_negatives)\n            \n            balanced_candidates = positives + negatives\n            \n            if len(balanced_candidates) > MAX_CANDIDATES_PER_SESSION:\n                negatives = random.sample(negatives, MAX_CANDIDATES_PER_SESSION - len(positives))\n                balanced_candidates = positives + negatives\n            \n            rows = compute_features_for_session(\n                input_aids, input_types, input_ts, balanced_candidates, event_type,\n                models[event_type], indexes[event_type], idx_to_aids[event_type], popular_items\n            )\n            \n            if not rows:\n                continue\n            \n            for row, aid in zip(rows, balanced_candidates):\n                row['label'] = 1 if aid in pos_set else 0\n            \n            df_features = pd.DataFrame(rows)\n            \n            for col in df_features.select_dtypes(include=['float64']).columns:\n                df_features[col] = pd.to_numeric(df_features[col], downcast='float')\n            for col in df_features.select_dtypes(include=['int64']).columns:\n                df_features[col] = pd.to_numeric(df_features[col], downcast='integer')\n            \n            if i < train_size:\n                train_chunks.append(df_features)\n            else:\n                val_chunks.append(df_features)\n\nsave_batch(batch_num, train_chunks, val_chunks)\n\nprint(f\"Всего батчей: {batch_num + 1}\")\n\nprint(f\"После фильтра (с carts И orders): {filtered_sessions}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-09T23:57:17.364475Z","iopub.status.idle":"2026-07-09T23:57:17.364908Z","shell.execute_reply.started":"2026-07-09T23:57:17.364684Z","shell.execute_reply":"2026-07-09T23:57:17.364710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### обучаемся по батчам чтобы не упасть по RAM и используем catboost pool","metadata":{}},{"cell_type":"code","source":"import catboost as cb\nimport glob\nimport pickle\nimport gc\n\ntrain_files = sorted(glob.glob('/kaggle/working/train_data_batch_*.parquet'))\nval_files = sorted(glob.glob('/kaggle/working/val_data_batch_*.parquet'))\n\nfeature_cols = None\ncat_features = ['event_type']\ntype_weights = {'clicks': 1.0, 'carts': 3.0, 'orders': 6.0}\n\nmodel = cb.CatBoostClassifier(\n    iterations=500,\n    depth=6,\n    learning_rate=0.05,\n    loss_function='Logloss',\n    eval_metric='AUC',\n    random_seed=11,\n    cat_features=cat_features,\n    verbose=100,\n    thread_count=-1,\n    l2_leaf_reg=3.0,\n    bagging_temperature=1.0,\n    subsample=0.8,\n    colsample_bylevel=0.8,\n    min_data_in_leaf=20,\n    border_count=128\n)\n\nprint(\"Загрузка val данных\")\nval_dfs = []\nfor f in val_files:\n    df = pd.read_parquet(f)\n    for col in df.select_dtypes(include=['float64']).columns:\n        df[col] = pd.to_numeric(df[col], downcast='float')\n    for col in df.select_dtypes(include=['int64']).columns:\n        df[col] = pd.to_numeric(df[col], downcast='integer')\n    val_dfs.append(df)\n\ndf_val = pd.concat(val_dfs, ignore_index=True)\ndel val_dfs\ngc.collect()\n\n\nfeature_cols = [c for c in df_val.columns if c not in ['label']]\n\nval_pool = cb.Pool(\n    df_val[feature_cols],\n    df_val['label'],\n    cat_features=cat_features,\n    weight=df_val['event_type'].map(type_weights).values\n)\ndel df_val\ngc.collect()\n\n\nprint(\"Обучение по батчам\")\nfor i, f in enumerate(train_files):\n    print(f\"Батч {i+1}/{len(train_files)}\")\n    df = pd.read_parquet(f)\n    \n    for col in df.select_dtypes(include=['float64']).columns:\n        df[col] = pd.to_numeric(df[col], downcast='float')\n    for col in df.select_dtypes(include=['int64']).columns:\n        df[col] = pd.to_numeric(df[col], downcast='integer')\n    \n    train_pool = cb.Pool(\n        df[feature_cols],\n        df['label'],\n        cat_features=cat_features,\n        weight=df['event_type'].map(type_weights).values\n    )\n    \n    if i == 0:\n        model.fit(\n            train_pool,\n            eval_set=val_pool,\n            early_stopping_rounds=50,\n            use_best_model=False,\n            plot=False\n        )\n    else:\n        model.fit(\n            train_pool,\n            eval_set=val_pool,\n            early_stopping_rounds=50,\n            use_best_model=False,\n            plot=False,\n            init_model=model\n        )\n    \n    del df, train_pool\n    \n    gc.collect()\n    \n\nprint(\"\\nОбучение завершено!\")\nprint(f\"Всего деревьев: {model.tree_count_}\")\n\nmodel.save_model('/kaggle/working/catboost_ranker.cbm')\nprint(f\"Модель сохранена\")\n\nwith open('/kaggle/working/feature_cols.pkl', 'wb') as f:\n    pickle.dump(feature_cols, f)\nprint(f\"Список фич сохранен\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T00:18:30.171655Z","iopub.execute_input":"2026-07-10T00:18:30.172598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import catboost as cb\nimport pickle\n\nmodel = cb.CatBoostClassifier()\nmodel.load_model('/kaggle/input/datasets/igorvorobev7892/catboost-model-for-recproj2/catboost_ranker.cbm')\n\nprint(f\"Модель загружена успешно!\")\n\nwith open('/kaggle/input/datasets/igorvorobev7892/catboost-model-for-recproj2/feature_cols.pkl', 'rb') as f:\n    feature_cols = pickle.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T13:10:54.668418Z","iopub.execute_input":"2026-07-14T13:10:54.668763Z","iopub.status.idle":"2026-07-14T13:10:57.769383Z","shell.execute_reply.started":"2026-07-14T13:10:54.668733Z","shell.execute_reply":"2026-07-14T13:10:57.768576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom collections import defaultdict\n\n\ndef get_recommendations(session_events, item2vec_models, indexes, idx_to_aids, popular_items,\n                        catboost_model, feature_cols, top_k=20):\n    input_aids = [str(e['aid']) for e in session_events]\n    input_types = [e['type'] for e in session_events]\n    input_ts = [float(e['ts']) for e in session_events]\n    \n    if not input_aids:\n        return {\n            'clicks': popular_items.get('clicks', [])[:top_k],\n            'carts': popular_items.get('carts', [])[:top_k],\n            'orders': popular_items.get('orders', [])[:top_k]\n        }\n    \n    recommendations = {}\n    \n    for event_type in ['clicks', 'carts', 'orders']:\n        candidates = generate_candidates(\n            input_aids, input_types, input_ts,\n            event_type, top_k=150, last_n=20\n        )\n        \n        rows = compute_features_for_session(\n            input_aids, input_types, input_ts, candidates, event_type,\n            item2vec_models[event_type], indexes[event_type], idx_to_aids[event_type], \n            popular_items\n        )\n        \n        if not rows:\n            recommendations[event_type] = popular_items.get(event_type, [])[:top_k]\n            continue\n        \n        df_features = pd.DataFrame(rows)\n        \n        X = df_features[feature_cols]\n        \n        scores = catboost_model.predict_proba(X)[:, 1]\n        \n        scored_candidates = list(zip(candidates, scores))\n        scored_candidates.sort(key=lambda x: x[1], reverse=True)\n        \n        top_candidates = [aid for aid, score in scored_candidates[:top_k]]\n        \n        recommendations[event_type] = top_candidates\n    \n    return recommendations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T13:30:37.369151Z","iopub.execute_input":"2026-07-14T13:30:37.370197Z","iopub.status.idle":"2026-07-14T13:30:37.380401Z","shell.execute_reply.started":"2026-07-14T13:30:37.370140Z","shell.execute_reply":"2026-07-14T13:30:37.379305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ujson as json\nimport numpy as np\nimport pandas as pd\nimport catboost as cb\nimport pickle\nfrom collections import defaultdict\nimport gc\n\nTRAIN_PATH = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\n\nrecalls_by_type = defaultdict(list)\nprocessed_sessions = 0\ntotal_lines = 0\n\nwith open(TRAIN_PATH) as f:\n    for i, line in enumerate(f):\n        if i+1 <= train_size:\n            continue\n        total_lines += 1\n        \n        record = json.loads(line)\n        events = record.get('events', [])\n        \n        # Делим сессию 80/20 как при обучении\n        split_idx = max(1, int(len(events) * 0.8))\n        input_events = events[:split_idx]\n        target_events = events[split_idx:]\n        \n        if not target_events or len(input_events) == 0:\n            continue\n        \n        recs = get_recommendations(\n            session_events=input_events,\n            item2vec_models=item2vec_models,\n            indexes=indexes,\n            idx_to_aids=idx_to_aids,\n            popular_items=popular_items,\n            catboost_model=model,\n            feature_cols=feature_cols,\n            top_k=20\n        )\n        \n        processed_sessions += 1\n        \n        true_positives = defaultdict(set)\n        for e in target_events:\n            true_positives[e['type']].add(str(e['aid']))\n        \n        for event_type in ['clicks', 'carts', 'orders']:\n            true_set = true_positives[event_type]\n            if not true_set:\n                continue\n                \n            pred_set = set(recs.get(event_type, []))\n            recall = len(pred_set & true_set) / len(true_set)\n            recalls_by_type[event_type].append(recall)\n        \n        if (i + 1) % 5000 == 0:\n            print(f\"  Обработано: {processed_sessions} сессий | \"\n                  f\"Clicks R@20={np.mean(recalls_by_type['clicks']):.4f} | \"\n                  f\"Carts R@20={np.mean(recalls_by_type['carts']):.4f} | \"\n                  f\"Orders R@20={np.mean(recalls_by_type['orders']):.4f}\")\n\n\nprint(f\"Всего прочитано строк: {total_lines}\")\nprint(f\"Обработано сессий: {processed_sessions}\")\n\nrecall_scores = {}\nfor event_type in ['clicks', 'carts', 'orders']:\n    r_list = recalls_by_type[event_type]\n    avg_recall = np.mean(r_list) if r_list else 0.0\n    recall_scores[event_type] = avg_recall\n    print(f\"{event_type}: recall@20 = {avg_recall:.4f} (сессий с таргетом: {len(r_list)})\")\n\nweighted_recall = (\n    0.10 * recall_scores['clicks'] +\n    0.30 * recall_scores['carts'] +\n    0.60 * recall_scores['orders']\n)\n\nprint(f\"Взвешенный RECALL@20 = {weighted_recall:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-14T13:38:29.259280Z","iopub.execute_input":"2026-07-14T13:38:29.260107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}