{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":3618498,"sourceType":"datasetVersion","datasetId":1931827},{"sourceId":94922230,"sourceType":"kernelVersion"}],"dockerImageVersionId":30153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 52nd Place Solution Notebook\n\nThis notebook is a cleaned version of my final submission.  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324076/) for some details about my solution.\n\nThe notebook has minimal code in it - most of the code is imported from my [handmhelpers dataset](https://www.kaggle.com/datasets/jacob34/handmhelpers), which is synced to [this github repo](https://github.com/JacobCP/kaggle-handm-helpers) .  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324078) for some details about my code development.  \n\n**Please note:**  \nI plan on continuing to update the github repo, as I try to recreate some of the strategies shared by winning teams.  \nSome of those changes may break the code usage for this notebook.  \nIn order to keep this notebook functional, I will no longer be updating the dataset to reflect the changes made to the repo - it will remain at commit 86c412e902a7692b24e15791322a8dfeb5a761eb","metadata":{}},{"cell_type":"code","source":"%%time\nimport os\nimport sys\nimport copy\nfrom datetime import datetime\nimport gc\nimport pickle as pkl\nimport shelve\n\nimport pandas as pd\nimport numpy as np\nimport cudf\n    \nsys.path.append(\"../input/\")\nfrom handmhelpers import io as h_io, sub as h_sub, cv as h_cv, fe as h_fe\nfrom handmhelpers import modeling as h_modeling, candidates as h_can, pairs as h_pairs","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:13.581442Z","iopub.execute_input":"2023-12-27T14:55:13.582037Z","iopub.status.idle":"2023-12-27T14:55:13.588304Z","shell.execute_reply.started":"2023-12-27T14:55:13.581994Z","shell.execute_reply":"2023-12-27T14:55:13.587481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and convert data","metadata":{}},{"cell_type":"code","source":"%%time\n# load dữ liệu từ các file csv, trong đó các trường trong file có sẵn được đặt lại kiểu dữ liệu\n# các trường _id, _no  đặt kiểu dữ liệu là \"int8, int16, int32\", các trường khác đặt kiểu dữ liệu là \"category\"\ncustomers, transactions, articles = h_io.load_data(files=['customers.csv', 'transactions_train.csv', 'articles.csv'])        \n# thay các giá trị id của customer_id thành 0,1,2,3,...(kiểu \"int32\") ở data_frame customers và transactions\nindex_to_id_dict_path = h_fe.reduce_customer_id_memory(customers, [transactions])\n#thêm trường \"week_number\" của data_frame transactions lưu giá trị là tuần thứ mấy của data frame, kiểu \"int8\"\ntransactions[\"week_number\"] = h_fe.day_week_numbers(transactions[\"t_dat\"])\n# data_frame transactions: thay các giá trị thời gian của trường t_dat thành ngày thứ mấy của data frame, kiểu \"int16\"\ntransactions[\"t_dat\"] = h_fe.day_numbers(transactions[\"t_dat\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:17.318376Z","iopub.execute_input":"2023-12-27T14:55:17.318660Z","iopub.status.idle":"2023-12-27T14:55:23.145526Z","shell.execute_reply.started":"2023-12-27T14:55:17.318626Z","shell.execute_reply":"2023-12-27T14:55:23.144539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:26.763928Z","iopub.execute_input":"2023-12-27T14:55:26.764658Z","iopub.status.idle":"2023-12-27T14:55:28.333988Z","shell.execute_reply.started":"2023-12-27T14:55:26.764622Z","shell.execute_reply":"2023-12-27T14:55:28.333234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:30.601784Z","iopub.execute_input":"2023-12-27T14:55:30.602548Z","iopub.status.idle":"2023-12-27T14:55:30.672002Z","shell.execute_reply.started":"2023-12-27T14:55:30.602514Z","shell.execute_reply":"2023-12-27T14:55:30.671163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get item pairs","metadata":{}},{"cell_type":"code","source":"%%time\n\npairs_per_item = 5\n\nweek_number_pairs = {}\n\n\"\"\"\nhàm create_pairs hoạt động như sau:\nvới tuần (week_number), có 2 dataframe là working_t_df, pairs_t_df lần lượt là danh sách các sản phẩm mua từ trước đến tuần đang xét(week_number) và danh sách các sp mua đúng vào tuần đang xét (week_number)\nchỉ lấy các trường customer_id và article_id, lọc ra các cặp customer_id và article_id trùng nhau\n\nlần lượt xét các batch_size, mỗi batch_size có 5000 sản phẩm khác nhau\nbatch_t lưu các sản phẩm trong batch đang xét từ working_t_df\nvới mỗi sản phẩm trong batch, thêm số lượng khách hàng đã mua (trừ đi 1, không tính bản thân khách hàng)\nmerge batch_t và pairs_t_df vào batch_pairs_df lưu các cặp sản phẩm của từng khách hàng, với một sản phẩm ở tuần week_number và tuần trước đó sao cho hai sản phẩm không trùng nhau\nxóa các cặp sản phẩm chỉ có một người mua\nsắp xếp các cặp sản phẩm có nhiều customer_id mua nhất rồi chọn ra 5 sản phẩm\nthêm trường percent_customers là thống kê phần trăm (được tính bằng số khách hàng mua cặp sản phẩm / số khách hàng mua sản phẩm đang xét)\n\nhàm trả về data frame lưu thông tin về article_id, pair_article_id, customer_count, percent_customers của các cặp sản phẩm tại tuần week_number\n\"\"\"\"\n\n# đoạn code trả về một list các data frame của nhiều tuần\nfor week_number in [96, 97, 98, 99, 100, 101, 102, 103, 104]:\n    print(f\"Creating pairs for week number {week_number}\")\n    week_number_pairs[week_number] = h_pairs.create_pairs(\n        transactions, week_number, pairs_per_item, verbose=False\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:07.472666Z","iopub.execute_input":"2023-12-27T14:55:07.473452Z","iopub.status.idle":"2023-12-27T14:55:07.486536Z","shell.execute_reply.started":"2023-12-27T14:55:07.473414Z","shell.execute_reply":"2023-12-27T14:55:07.485500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_number = 96\nworking_t_df = transactions[[\"customer_id\", \"article_id\", \"week_number\"]].copy()\n\n# we'll look for pairs in last week only\nworking_t_df = working_t_df.query(f\"week_number <= {week_number}\").copy()\npairs_t_df = working_t_df.query(f\"week_number == {week_number}\").copy()\npairs_t_df.columns = [\"customer_id\", \"pair_article_id\", \"week_number\"]\n\n# drop week number, and drop duplicates\ndel working_t_df[\"week_number\"], pairs_t_df[\"week_number\"]\nworking_t_df = working_t_df.drop_duplicates()\npairs_t_df = pairs_t_df.drop_duplicates()\n\n# setup the loop\nunique_articles = working_t_df[\"article_id\"].unique()\nbatch_size = 5000\nbatch_pairs_dfs = []\n","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:35.669617Z","iopub.execute_input":"2023-12-27T14:55:35.670343Z","iopub.status.idle":"2023-12-27T14:55:35.973471Z","shell.execute_reply.started":"2023-12-27T14:55:35.670303Z","shell.execute_reply":"2023-12-27T14:55:35.972638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_articles = unique_articles[0 : 0 + batch_size]\nbatch_t = working_t_df[working_t_df[\"article_id\"].isin(batch_articles)]\nbatch_t","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:38.413133Z","iopub.execute_input":"2023-12-27T14:55:38.413425Z","iopub.status.idle":"2023-12-27T14:55:38.646066Z","shell.execute_reply.started":"2023-12-27T14:55:38.413392Z","shell.execute_reply":"2023-12-27T14:55:38.645189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nhóm các articles, rồi đếm các customer của mỗi article, \nall_cust_counts = batch_t.groupby(\"article_id\")[\"customer_id\"].nunique()\nprint(all_cust_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:40.989898Z","iopub.execute_input":"2023-12-27T14:55:40.990190Z","iopub.status.idle":"2023-12-27T14:55:41.021489Z","shell.execute_reply.started":"2023-12-27T14:55:40.990158Z","shell.execute_reply":"2023-12-27T14:55:41.020705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cust_counts = all_cust_counts.reset_index()\nprint(all_cust_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:43.021365Z","iopub.execute_input":"2023-12-27T14:55:43.021669Z","iopub.status.idle":"2023-12-27T14:55:43.045089Z","shell.execute_reply.started":"2023-12-27T14:55:43.021634Z","shell.execute_reply":"2023-12-27T14:55:43.044360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cust_counts.columns = [\"article_id\", \"all_customer_counts\"]\nprint(all_cust_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:45.408287Z","iopub.execute_input":"2023-12-27T14:55:45.408589Z","iopub.status.idle":"2023-12-27T14:55:45.431020Z","shell.execute_reply.started":"2023-12-27T14:55:45.408556Z","shell.execute_reply":"2023-12-27T14:55:45.430102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cust_counts[\"all_customer_counts\"] -= 1  # trừ sản phẩm mà khách hàng mua ở đúng tuần week_number\nprint(all_cust_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:47.989945Z","iopub.execute_input":"2023-12-27T14:55:47.990711Z","iopub.status.idle":"2023-12-27T14:55:48.012346Z","shell.execute_reply.started":"2023-12-27T14:55:47.990671Z","shell.execute_reply":"2023-12-27T14:55:48.011385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_pairs_df = batch_t.merge(pairs_t_df, on=\"customer_id\")\nprint(batch_pairs_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:50.141539Z","iopub.execute_input":"2023-12-27T14:55:50.141840Z","iopub.status.idle":"2023-12-27T14:55:50.171636Z","shell.execute_reply.started":"2023-12-27T14:55:50.141793Z","shell.execute_reply":"2023-12-27T14:55:50.170867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xóa các dòng mà cặp sản phẩm có cùng id\n# sản phẩm có id mua trc tuần đang xét vs id ở mua tuần xét ko được trùng nhau\nsame_article_row_idxs = batch_pairs_df.query(\n    \"article_id==pair_article_id\"\n).index\nbatch_pairs_df = batch_pairs_df.drop(same_article_row_idxs)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:52.092371Z","iopub.execute_input":"2023-12-27T14:55:52.092633Z","iopub.status.idle":"2023-12-27T14:55:52.115450Z","shell.execute_reply.started":"2023-12-27T14:55:52.092603Z","shell.execute_reply":"2023-12-27T14:55:52.114815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete single customer articles\nc1s = (\n    batch_pairs_df.groupby(\"article_id\")[[\"customer_id\"]]\n    .nunique()\n    .query(\"customer_id==1\")\n    .index\n)\nsingle_customer_row_idxs = batch_pairs_df[\n    batch_pairs_df[\"article_id\"].isin(c1s)\n].index\nbatch_pairs_df = batch_pairs_df.drop(single_customer_row_idxs)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:54.605256Z","iopub.execute_input":"2023-12-27T14:55:54.605553Z","iopub.status.idle":"2023-12-27T14:55:54.657592Z","shell.execute_reply.started":"2023-12-27T14:55:54.605525Z","shell.execute_reply":"2023-12-27T14:55:54.656983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get sorted counts of article-pair occurences\nbatch_pairs_df = batch_pairs_df.groupby([\"article_id\", \"pair_article_id\"])[\n    [\"customer_id\"]\n].count()\nbatch_pairs_df.columns = [\"customer_count\"]\nbatch_pairs_df = batch_pairs_df.reset_index()\nbatch_pairs_df = batch_pairs_df.sort_values(\n    [\"article_id\", \"customer_count\"], ascending=False\n)\nbatch_pairs_df","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:55:56.566294Z","iopub.execute_input":"2023-12-27T14:55:56.566971Z","iopub.status.idle":"2023-12-27T14:55:56.650230Z","shell.execute_reply.started":"2023-12-27T14:55:56.566929Z","shell.execute_reply":"2023-12-27T14:55:56.649507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cudf_groupby_head(df, groupby, head_count):\n    df = df.to_pandas()\n\n    head_df = df.groupby(groupby).head(head_count)\n\n    head_df = cudf.DataFrame(head_df)\n\n    return head_df","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:56:00.340837Z","iopub.execute_input":"2023-12-27T14:56:00.341579Z","iopub.status.idle":"2023-12-27T14:56:00.346035Z","shell.execute_reply.started":"2023-12-27T14:56:00.341544Z","shell.execute_reply":"2023-12-27T14:56:00.345281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pairs_per_item = 5\n# get top x pairs for each article\nbatch_pairs_df = cudf_groupby_head(batch_pairs_df, \"article_id\", pairs_per_item)\nbatch_pairs_df","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:56:05.594802Z","iopub.execute_input":"2023-12-27T14:56:05.595457Z","iopub.status.idle":"2023-12-27T14:56:05.699905Z","shell.execute_reply.started":"2023-12-27T14:56:05.595421Z","shell.execute_reply":"2023-12-27T14:56:05.698964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate percentage statistic\nbatch_pairs_df = batch_pairs_df.merge(all_cust_counts, on=\"article_id\")\nbatch_pairs_df[\"percent_customers\"] = (\n    batch_pairs_df[\"customer_count\"] / batch_pairs_df[\"all_customer_counts\"]\n)\ndel batch_pairs_df[\"all_customer_counts\"]\n\nbatch_pairs_dfs.append(batch_pairs_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:56:12.861300Z","iopub.execute_input":"2023-12-27T14:56:12.861603Z","iopub.status.idle":"2023-12-27T14:56:12.870062Z","shell.execute_reply.started":"2023-12-27T14:56:12.861568Z","shell.execute_reply":"2023-12-27T14:56:12.869295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_number_pairs = {}\nweek_number_pairs[96] = h_pairs.create_pairs(\n        transactions, 96, pairs_per_item, verbose=False\n    )\nweek_number_pairs[96]","metadata":{"execution":{"iopub.status.busy":"2023-12-27T14:56:21.758163Z","iopub.execute_input":"2023-12-27T14:56:21.758468Z","iopub.status.idle":"2023-12-27T14:56:29.082431Z","shell.execute_reply.started":"2023-12-27T14:56:21.758436Z","shell.execute_reply":"2023-12-27T14:56:29.081678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Main retrieval/features function!","metadata":{}},{"cell_type":"code","source":"def create_candidates_with_features_df(t, c, a, customer_batch=None, **kwargs):\n    # splitting cv\n    features_df, label_df = h_cv.feature_label_split(\n        t, kwargs[\"label_week\"], kwargs[\"feature_periods\"]\n    )\n    \n    # converting relative day_number\n    features_df[\"t_dat\"] = h_fe.how_many_ago(features_df[\"t_dat\"])\n    features_df[\"week_number\"] = h_fe.how_many_ago(features_df[\"week_number\"])\n    \n    # pull out the cv week\n    article_pairs_df = week_number_pairs[kwargs[\"label_week\"]-1]\n    \n    # check if we can limit customers\n    if len(label_df) > 0:\n        customers = label_df[\"customer_id\"].unique()\n    elif customer_batch is not None:\n        customers = customer_batch\n    else:\n        customers = None\n    \n    ############################################\n    # creating candidates (and adding features)\n    ###########################################\n    \n    features_db = shelve.open(\"features_db\") \n    \n    # creating candidate (and saving features created)\n    recent_customer_cand, features_db[\"customer_article\"] = (\n        h_can.create_recent_customer_candidates(\n            features_df,\n            kwargs[\"ca_num_weeks\"],\n            customers=customers,\n        )\n    )\n    \n    (cust_last_week_cand,\n     cust_last_week_pair_cand,\n     features_db[\"clw\"],\n     features_db[\"clw_pairs\"]) = h_can.create_last_customer_weeks_and_pairs(\n        features_df,\n        article_pairs_df,\n        kwargs[\"clw_num_weeks\"],\n        kwargs[\"clw_num_pair_weeks\"],\n        customers=customers,\n    )\n    \n    _, features_db[\"popular_articles\"] = h_can.create_popular_article_cand(\n        features_df,\n        c,\n        a,\n        kwargs[\"pa_num_weeks\"],\n        kwargs[\"hier_col\"],\n        num_candidates=kwargs[\"num_recent_candidates\"],\n        num_articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    age_bucket_can, _, _ = h_can.create_age_bucket_candidates(\n        features_df,\n        c,\n        kwargs[\"num_age_buckets\"],\n        articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    \n    cand = [recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can]\n    cand = cudf.concat(cand).drop_duplicates()\n    cand = cand.sort_values([\"customer_id\", \"article_id\"]).reset_index(drop=True)\n    \n    del recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can\n    \n    cand = h_can.filter_candidates(cand, t, **kwargs)\n    \n    # creating other features\n    h_fe.create_cust_hier_features(features_df, a, kwargs[\"hier_cols\"], features_db)\n    h_fe.create_price_features(features_df, features_db)\n    h_fe.create_cust_features(c, features_db)\n    h_fe.create_article_cust_features(features_df, c, features_db)\n    h_fe.create_lag_features(features_df, a, kwargs[\"lag_days\"], features_db)\n    h_fe.create_rebuy_features(features_df, features_db)\n    h_fe.create_cust_t_features(features_df, a, features_db)\n    h_fe.create_art_t_features(features_df, features_db)\n    \n    del features_df\n\n    # another filter at the end, for the ones that didn't get filtered earlier\n    if customers is not None:\n        cand = cand[cand[\"customer_id\"].isin(customers)]\n    \n    # report on recall/precision of candidates\n    if kwargs[\"cv\"]:\n        ground_truth_candidates = label_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n        h_cv.report_candidates(cand, ground_truth_candidates)\n        del ground_truth_candidates        \n    \n    # adding features to candidates\n    cand_with_f_df = h_can.add_features_to_candidates(\n        cand, features_db, c, a\n    )\n    \n    # manually adding article features (couldn't use shelve for some reason)\n    for article_col in kwargs[\"article_columns\"]:\n        art_col_map = a.set_index(\"article_id\")[article_col]\n        cand_with_f_df[article_col] = cand_with_f_df[\"article_id\"].map(art_col_map)\n    \n    # limiting features\n    if kwargs[\"selected_features\"] is not None:\n        cand_with_f_df = cand_with_f_df[\n            [\"customer_id\", \"article_id\"] + kwargs[\"selected_features\"]\n        ]\n        \n    features_db.close()\n    os.remove(\"features_db.bak\"), os.remove(\"features_db.dir\"), os.remove(\"features_db.dat\")\n    \n    assert len(cand) == len(cand_with_f_df), \"seem to have duplicates in the feature dfs\"\n    del cand\n    \n    return cand_with_f_df, label_df","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:00:18.347408Z","iopub.execute_input":"2022-05-10T18:00:18.347953Z","iopub.status.idle":"2022-05-10T18:00:18.368451Z","shell.execute_reply.started":"2022-05-10T18:00:18.347918Z","shell.execute_reply":"2022-05-10T18:00:18.367536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_model_score(ids_df, preds, truth_df):\n    predictions = h_modeling.create_predictions(ids_df, preds)\n    true_labels = h_cv.ground_truth(truth_df).set_index(\"customer_id\")[\"prediction\"]\n    score = round(h_cv.comp_average_precision(true_labels, predictions),5)\n    \n    return score","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:00:21.274676Z","iopub.execute_input":"2022-05-10T18:00:21.276049Z","iopub.status.idle":"2022-05-10T18:00:21.282617Z","shell.execute_reply.started":"2022-05-10T18:00:21.276Z","shell.execute_reply":"2022-05-10T18:00:21.281939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parameters - one place for all!","metadata":{}},{"cell_type":"code","source":"cv_params = {\n    \"cv\": True,\n    \"feature_periods\": 105,\n    \"label_week\": 104,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 36,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\"n_estimators\": 200, \"num_leaves\": 20},\n    \"log_evaluation\": 10,\n    \"early_stopping\": 20,\n    \"eval_at\": 12,\n    \"save_model\": True,\n    \"num_concats\": 5,\n}\nsub_params = {\n    \"cv\": False,\n    \"feature_periods\": 105,\n    \"label_week\": 105,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 60,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\n        \"n_estimators\": 150,\n        \"num_leaves\": 20,    \n    },\n    \"log_evaluation\": 10,\n    \"eval_at\": 12,\n    \"prediction_models\": [\"model_104\", \"model_105\"],\n    \"save_model\": True,\n    \"num_concats\": 5,\n}","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:01:00.324592Z","iopub.execute_input":"2022-05-10T18:01:00.324852Z","iopub.status.idle":"2022-05-10T18:01:00.334824Z","shell.execute_reply.started":"2022-05-10T18:01:00.324823Z","shell.execute_reply":"2022-05-10T18:01:00.334123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cand_features_func = create_candidates_with_features_df\nscoring_func = calculate_model_score","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:01:01.429553Z","iopub.execute_input":"2022-05-10T18:01:01.430098Z","iopub.status.idle":"2022-05-10T18:01:01.433894Z","shell.execute_reply.started":"2022-05-10T18:01:01.430047Z","shell.execute_reply":"2022-05-10T18:01:01.433039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncv_weeks = [104]\nresults = h_modeling.run_all_cvs(\n    t, c, a, cand_features_func, scoring_func, \n    cv_weeks=cv_weeks, **cv_params\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:01:02.390339Z","iopub.execute_input":"2022-05-10T18:01:02.390591Z","iopub.status.idle":"2022-05-10T18:02:42.160113Z","shell.execute_reply.started":"2022-05-10T18:01:02.390564Z","shell.execute_reply":"2022-05-10T18:02:42.159372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ngc.collect()\nh_modeling.full_sub_train_run(t, c, a, cand_features_func, scoring_func, **sub_params)\npredictions = h_modeling.full_sub_predict_run(\n    t, c, a, cand_features_func, **sub_params\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:05:55.592758Z","iopub.execute_input":"2022-05-10T18:05:55.593023Z","iopub.status.idle":"2022-05-10T18:13:51.253149Z","shell.execute_reply.started":"2022-05-10T18:05:55.592994Z","shell.execute_reply":"2022-05-10T18:13:51.252409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = h_sub.create_sub(c[\"customer_id\"], predictions, index_to_id_dict_path)\nsub.to_csv('dev_submission.csv', index=False)\n\ndisplay(sub.head())\nprint(sub.shape)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T18:16:30.682484Z","iopub.execute_input":"2022-05-10T18:16:30.68274Z","iopub.status.idle":"2022-05-10T18:18:14.139011Z","shell.execute_reply.started":"2022-05-10T18:16:30.68271Z","shell.execute_reply":"2022-05-10T18:18:14.137713Z"},"trusted":true},"execution_count":null,"outputs":[]}]}