{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%time\nimport os\nimport sys\nimport copy\nfrom datetime import datetime\nimport gc\nimport pickle as pkl\nimport shelve\n\nimport pandas as pd\nimport numpy as np\nimport cudf\nimport itertools\nfrom typing import Union, List\n\nimport lightgbm\nfrom lightgbm import LGBMRanker\nfrom sklearn.metrics import roc_auc_score\n\nimport math\nfrom datetime import timedelta\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:32:32.869712Z","iopub.execute_input":"2023-02-21T04:32:32.870125Z","iopub.status.idle":"2023-02-21T04:32:37.444287Z","shell.execute_reply.started":"2023-02-21T04:32:32.870047Z","shell.execute_reply":"2023-02-21T04:32:37.442710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load and convert data","metadata":{}},{"cell_type":"code","source":"dtypes = {\n    # articles\n    \"article_id\": \"int32\",\n    \"prod_name\": \"category\",\n    \"product_type_name\": \"category\",\n    \"product_group_name\": \"category\",\n    \"graphical_appearance_name\": \"category\",\n    \"colour_group_name\": \"category\",\n    \"perceived_colour_value_name\": \"category\",\n    \"perceived_colour_master_name\": \"category\",\n    \"department_name\": \"category\",\n    \"garment_group_name\": \"category\",\n    \"section_name\": \"category\",\n    \"index_group_name\": \"category\",\n    \"index_code\": \"category\",\n    \"index_name\": \"category\",\n    \"detail_desc\": \"category\",\n    \"product_code\": \"int32\",\n    \"product_type_no\": \"int16\",\n    \"graphical_appearance_no\": \"int32\",\n    \"colour_group_code\": \"int8\",\n    \"perceived_colour_value_id\": \"int8\",\n    \"perceived_colour_master_id\": \"int8\",\n    \"department_no\": \"int16\",\n    \"index_group_no\": \"int8\",\n    \"section_no\": \"int8\",\n    \"garment_group_no\": \"int16\",\n    # customers\n    \"FN\": \"category\",\n    \"Active\": \"category\",\n    \"club_member_status\": \"category\",\n    \"fashion_news_frequency\": \"category\",\n    \"age\": \"float32\",\n    \"postal_code\": \"category\",\n    # transactions\n    \"price\": \"float32\",\n    \"sales_channel_id\": \"int8\",\n}\n\n\ncompetition_directory = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/\"\nfile_names = [\n    \"articles.csv\",\n    \"customers.csv\",\n    \"sample_submission.csv\",\n    \"transactions_train.csv\",\n]\n\n\ndef load_data(data_path=competition_directory, files=file_names):\n    loaded_dfs = []\n    for file_name in files:\n        file_path = os.path.join(data_path, file_name)\n        df = cudf.read_csv(file_path)\n        for column in df:\n            if column in dtypes:\n                df[column] = df[column].astype(dtypes[column])\n        loaded_dfs.append(df)\n\n    return loaded_dfs","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:32:37.446626Z","iopub.execute_input":"2023-02-21T04:32:37.447336Z","iopub.status.idle":"2023-02-21T04:32:37.456872Z","shell.execute_reply.started":"2023-02-21T04:32:37.447296Z","shell.execute_reply":"2023-02-21T04:32:37.456027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_customer_id_memory(customers_df, other_dfs):\n    \"\"\"\n    reduces memory in dfs in place\n\n    returns path to pickled customer_id mapping, for submission use\n    \"\"\"\n    ####\n    # customer id compression\n\n    # save the mapping we'll need to get back\n    index_to_id_dict = customers_df[\"customer_id\"]\n    file_path = \"id_to_index_dict.pkl\"\n    with open(file_path, \"wb\") as f:\n        pkl.dump(index_to_id_dict, f)\n    del index_to_id_dict\n\n    # now compress\n    id_to_index_dict = customers_df.reset_index().set_index(\"customer_id\")[\"index\"]\n    customers_df[\"customer_id\"] = (\n        customers_df[\"customer_id\"].map(id_to_index_dict).astype(\"int32\")\n    )\n\n    for other_df in other_dfs:\n        if \"customer_id\" in other_df:\n            other_df[\"customer_id\"] = (\n                other_df[\"customer_id\"].map(id_to_index_dict).astype(\"int32\")\n            )\n\n    return file_path\n\n\ndef day_numbers(dates: cudf.Series):\n    \"\"\"\n    assign consecutive number to dates, such that earliest date is 0\n\n    args:\n    dates: pd.Series of date strings\n\n    returns:\n    all_day_numbers: pd.Series of day numbers each day in original pd.Series corresponds to\n\n    \"\"\"\n\n    unique_dates = dates.unique()\n    unique_dates = unique_dates.to_pandas()\n    unique_dates = np.sort(unique_dates)\n    number_range = np.arange(len(unique_dates))\n    date_number_dict = dict(zip(unique_dates, number_range))\n\n    all_day_numbers = dates.map(date_number_dict)\n    all_day_numbers = all_day_numbers.astype(\"int16\")\n\n    return all_day_numbers\n\n\ndef day_week_numbers(dates: cudf.Series):\n    \"\"\"\n    assign week numbers to dates, such that:\n    - week numbers represent consecutive actual weeks on the calendar\n    - the latest date in the dates provided is the last day of the latest week number\n\n    args:\n    dates: pd.Series of date strings\n\n    returns:\n    day_weeks: pd.Series of week numbers each day in original pd.Series is in\n\n    \"\"\"\n    pd_dates = cudf.to_datetime(dates)\n\n    unique_dates = cudf.Series(pd_dates.unique())\n    numbered_days = unique_dates - unique_dates.min() + timedelta(1)\n    numbered_days = numbered_days.dt.days\n    extra_days = numbered_days.max() % 7\n    numbered_days -= extra_days\n    day_weeks = (numbered_days / 7).applymap(lambda x: math.ceil(x))\n    day_weeks_map = cudf.DataFrame(\n        {\"day_weeks\": day_weeks, \"unique_dates\": unique_dates}\n    ).set_index(\"unique_dates\")[\"day_weeks\"]\n    all_day_weeks = pd_dates.map(day_weeks_map)\n    all_day_weeks = all_day_weeks.astype(\"int8\")\n\n    return all_day_weeks\n\n\ndef year_week_numbers(weeks: cudf.Series):\n    \"\"\"\n    convert consecutive week numbers into year and week features, such that:\n    - each year has 52 weeks\n    - the last week number is the series provided will be week 52 of the last year\n\n    e.g. week number 106 will be week 2 of year 3\n\n    args:\n    weeks: pd.Series of integers, representing consecutive weeks\n\n    returns:\n    years: integers representing consecutive \"years\" of 52 weeks\n    year_weeks: integers representing weeks within said \"years\"\n    \"\"\"\n\n    years = (weeks / 52).apply(math.ceil)\n    year_weeks = (weeks % 52).replace(0, 52)\n\n    return years, year_weeks\n\n\ndef how_many_ago(sequential_numbers: cudf.Series):\n    \"\"\"\n    given a pd_or_cudf.series of numbers between 0 and n,\n    returns new series with n subtracted from each element\n\n    (so that if n reflects last day or week number, output will reflect how many days or weeks\n    earlier each element was from last day or week, with -1 meaning 1 day or week earlier etc.)\n    \"\"\"\n\n    return sequential_numbers - sequential_numbers.max()\n\n\ndef create_cust_hier_features(transactions_df, articles_df, hier_cols, features_db):\n    sample_col = \"t_dat\"\n\n    # create hiers\n    for hier_col in hier_cols:\n        # total customer counts\n        total_cust_counts = transactions_df.groupby(\"customer_id\")[sample_col].count()\n\n        # add hierarchy column to transactions\n        article_hier_lookup = articles_df.set_index(\"article_id\")[hier_col]\n        transactions_df[hier_col] = transactions_df[\"article_id\"].map(\n            article_hier_lookup\n        )\n\n        # get customer/hierarchy statistics\n        cust_hier = (\n            transactions_df.groupby([\"customer_id\", hier_col])[sample_col]\n            .count()\n            .reset_index()\n        )\n        cust_hier.columns = list(cust_hier.columns)[:-1] + [\"cust_hier_counts\"]\n        cust_hier = cust_hier.sort_values(\n            [\"customer_id\", \"cust_hier_counts\"], ascending=False\n        )\n\n        cust_hier[\"total_counts\"] = cust_hier[\"customer_id\"].map(total_cust_counts)\n\n        hier_portion_column = f\"cust_{hier_col}_portion\"\n        cust_hier[hier_portion_column] = (\n            cust_hier[\"cust_hier_counts\"] / cust_hier[\"total_counts\"]\n        )\n        cust_hier = cust_hier[[\"customer_id\", hier_col, hier_portion_column]]\n        cust_hier = cust_hier.set_index([\"customer_id\", hier_col])\n        cust_hier[hier_portion_column] = cust_hier[hier_portion_column].astype(\n            \"float32\"\n        )\n        features_db[hier_portion_column] = ([\"customer_id\", hier_col], cust_hier)\n\ndef create_art_features(article_df, features_db):\n    atf_df = article_df[[\"article_id\",\"colour_group_code\",\"perceived_colour_value_id\", \"perceived_colour_master_id\", \"department_no\"]]\n    \n    features_db[\"art_features\"] = ([\"article_id\"], atf_df)\n    \ndef create_price_features(transactions_df, features_db):\n    ###################\n    # article_prices\n    ###################\n    article_prices_df = transactions_df.groupby(\"article_id\")[[\"price\"]].max()\n    article_prices_df.columns = [\"max_price\"]\n\n    last_week = transactions_df[\"week_number\"].max()\n    last_week_t_df = transactions_df.query(f\"week_number == {last_week}\")\n    last_week_prices = transactions_df.groupby(\"article_id\")[\"price\"].mean()\n    article_prices_df[\"last_week_price\"] = last_week_prices\n\n    article_prices_df = article_prices_df.dropna()\n\n    article_prices_df[\"last_week_price_ratio\"] = (\n        article_prices_df[\"last_week_price\"] / article_prices_df[\"max_price\"]\n    )\n\n    features_db[\"article_prices\"] = ([\"article_id\"], article_prices_df)\n\n    ############################\n    # customer price features\n    ############################\n    cust_prices_df = transactions_df[\n        [\"customer_id\", \"article_id\", \"week_number\", \"price\"]\n    ].copy()\n\n    cust_prices_df[\"max_article_price\"] = cust_prices_df[\"article_id\"].map(\n        cust_prices_df.groupby([\"article_id\"])[\"price\"].max()\n    )\n\n    # for each purchase, the previous article/week price, and price discount\n    article_week_price_df = (\n        cust_prices_df.groupby([\"week_number\", \"article_id\"])[\"price\"]\n        .mean()\n        .reset_index()\n    )\n    article_week_price_df.columns = [\n        \"week_number\",\n        \"article_id\",\n        \"article_previous_week_price\",\n    ]\n    article_week_price_df[\n        \"week_number\"\n    ] += 1  # for the next week, the price is from the previous week\n    cust_prices_df = cust_prices_df.merge(\n        article_week_price_df, on=[\"week_number\", \"article_id\"]\n    )\n    cust_prices_df[\"article_previous_week_price_ratio\"] = (\n        cust_prices_df[\"article_previous_week_price\"]\n        / cust_prices_df[\"max_article_price\"]\n    )\n    cust_prices_df = cust_prices_df.groupby(\"customer_id\")[\n        [\n            \"max_article_price\",\n            \"article_previous_week_price\",\n            \"article_previous_week_price_ratio\",\n        ]\n    ].mean()\n    cust_prices_df.columns = [\n        \"cust_avg_max_price\",\n        \"cust_avg_last_week_price\",\n        \"cust_avg_last_week_price_ratio\",\n    ]\n    cust_prices_df = cust_prices_df.dropna()\n\n    features_db[\"cust_price_features\"] = ([\"customer_id\"], cust_prices_df)\n\n\ndef create_cust_t_features(transactions_df, a, features_db):\n    ctf_df = transactions_df.groupby(\"customer_id\")[[\"sales_channel_id\"]].mean()\n    ctf_df.columns = [\"cust_sales_channel\"]\n    ctf_df[\"cust_sales_channel\"] = ctf_df[\"cust_sales_channel\"].astype(\"float32\")\n    ctf_df[\"cust_sales_channel\"] = ctf_df[\"cust_sales_channel\"].round(2) - 1.0\n\n    ctf_df[\"cust_t_counts\"] = (\n        transactions_df.groupby(\"customer_id\")[\"sales_channel_id\"]\n        .count()\n        .astype(\"float32\")\n    )\n    ctf_df[\"cust_u_t_counts\"] = (\n        transactions_df.groupby(\"customer_id\")[\"article_id\"].nunique().astype(\"float32\")\n    )\n\n    sub_t_df = transactions_df[[\"customer_id\", \"article_id\"]].copy()\n    sub_t_df[\"index_group_name\"] = sub_t_df[\"article_id\"].map(\n        a.set_index(\"article_id\")[\"index_group_name\"]\n    )\n    gender_dict = {\n        \"Ladieswear\": 1,\n        \"Baby/Children\": 0.5,\n        \"Menswear\": 0,\n        \"Sport\": 0.5,\n        \"Divided\": 0.5,\n    }\n    sub_t_df[\"article_gender\"] = (\n        sub_t_df[\"index_group_name\"].astype(\"str\").map(gender_dict)\n    )\n\n    sub_t_df[\"section_name\"] = sub_t_df[\"article_id\"].map(\n        a.set_index(\"article_id\")[\"section_name\"].astype(str)\n    )\n    sub_t_df.loc[sub_t_df[\"section_name\"] == \"Ladies H&M Sport\", \"article_gender\"] = 1\n    sub_t_df.loc[sub_t_df[\"section_name\"] == \"Men H&M Sport\", \"article_gender\"] = 0\n    ctf_df[\"cust_gender\"] = sub_t_df.groupby(\"customer_id\")[\"article_gender\"].mean()\n\n    features_db[\"cust_t_features\"] = ([\"customer_id\"], ctf_df)\n\n\ndef create_art_t_features(transactions_df, features_db):\n    atf_df = transactions_df.groupby(\"article_id\")[[\"sales_channel_id\"]].mean()\n    atf_df.columns = [\"art_sales_channel\"]\n    atf_df[\"art_sales_channel\"] = atf_df[\"art_sales_channel\"].astype(\"float32\")\n    atf_df[\"art_sales_channel\"] = atf_df[\"art_sales_channel\"].round(2) - 1.0\n\n    atf_df[\"art_t_counts\"] = (\n        transactions_df.groupby(\"article_id\")[\"sales_channel_id\"]\n        .count()\n        .astype(\"float32\")\n    )\n    atf_df[\"art_u_t_counts\"] = (\n        transactions_df.groupby(\"article_id\")[\"customer_id\"].nunique().astype(\"float32\")\n    )\n\n    features_db[\"art_t_features\"] = ([\"article_id\"], atf_df)\n\n\ndef create_cust_features(customers_df, features_db):\n    cust_df = customers_df.set_index(\"customer_id\")[[\"age\"]]\n    cust_df[\"age\"] = cust_df[\"age\"].astype(\"int16\").fillna(-1)\n\n    features_db[\"cust_features\"] = ([\"customer_id\"], cust_df)\n\n\ndef create_article_cust_features(transactions_df, customers_df, features_db):\n    art_cust_df = transactions_df[[\"article_id\", \"customer_id\"]].drop_duplicates()\n    cust_age = customers_df.set_index(\"customer_id\")[\"age\"]\n    art_cust_df[\"cust_age\"] = art_cust_df[\"customer_id\"].map(cust_age)\n    art_cust_df = art_cust_df.groupby(\"article_id\")[[\"cust_age\"]].mean()\n    art_cust_df.columns = [\"art_cust_age\"]\n\n    art_cust_df[\"art_cust_age\"] = art_cust_df[\"art_cust_age\"].astype(\"int16\").fillna(-1)\n\n    features_db[\"article_customer_age\"] = ([\"article_id\"], art_cust_df)\n\n\ndef create_lag_features(transactions_df, articles_df, lag_days, features_db):\n    last_date = transactions_df[\"t_dat\"].max()\n\n    article_counts_df = cudf.DataFrame(index=articles_df[\"article_id\"])\n    for lag_day in lag_days:\n        # column name\n        col_name = f\"last_{lag_day}_days_count\"\n\n        # column values\n        t_df_filtered = transactions_df[\n            transactions_df[\"t_dat\"] > (last_date - lag_day)\n        ]\n        lag_values = t_df_filtered.groupby(\"article_id\")[\"customer_id\"].nunique()\n\n        # putting them in\n        article_counts_df[col_name] = lag_values\n        article_counts_df[col_name] = article_counts_df[col_name]\n        article_counts_df[col_name] = article_counts_df[col_name].astype(\"float32\")\n\n    features_db[\"article_counts\"] = ([\"article_id\"], article_counts_df)\n\n\ndef create_rebuy_features(transactions_df, features_db):\n    duplicate_counts = transactions_df.groupby([\"customer_id\", \"article_id\"])[\n        \"week_number\"\n    ].count()\n    duplicate_counts = duplicate_counts.sort_values().reset_index()\n    duplicate_counts.columns = [\"customer_id\", \"article_id\", \"buy_count\"]\n    rebuy_ratio_df = duplicate_counts.groupby(\"article_id\")[[\"buy_count\"]].mean()\n    rebuy_ratio_df.columns = [\"rebuy_count_ratio\"]\n    rebuy_ratio_df[\"rebuy_count_ratio\"] = rebuy_ratio_df[\"rebuy_count_ratio\"].astype(\n        \"float32\"\n    )\n    duplicate_counts[\"buy_count\"] = duplicate_counts[\"buy_count\"].replace(1, 0)\n    duplicate_counts[\"buy_count\"][duplicate_counts[\"buy_count\"] > 1] = 1\n    rebuy_ratio_df[\"rebuy_ratio\"] = duplicate_counts.groupby(\"article_id\")[\n        \"buy_count\"\n    ].mean()\n\n    features_db[\"rebuy_features\"] = ([\"article_id\"], rebuy_ratio_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:32:37.460682Z","iopub.execute_input":"2023-02-21T04:32:37.460943Z","iopub.status.idle":"2023-02-21T04:32:37.503775Z","shell.execute_reply.started":"2023-02-21T04:32:37.460919Z","shell.execute_reply":"2023-02-21T04:32:37.502573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nc, t, a = load_data(files=['customers.csv', 'transactions_train.csv', 'articles.csv'])        \n\nindex_to_id_dict_path = reduce_customer_id_memory(c, [t])\nt[\"week_number\"] = day_week_numbers(t[\"t_dat\"])\nt[\"t_dat_tmp\"] = cudf.to_datetime(t['t_dat'])\nt[\"t_dat\"] = day_numbers(t[\"t_dat\"])","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:32:37.506802Z","iopub.execute_input":"2023-02-21T04:32:37.507497Z","iopub.status.idle":"2023-02-21T04:33:29.919967Z","shell.execute_reply.started":"2023-02-21T04:32:37.507459Z","shell.execute_reply":"2023-02-21T04:33:29.918909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntmp = t[['t_dat_tmp']].copy().to_pandas()\ntmp['dow'] = tmp['t_dat_tmp'].dt.dayofweek\ntmp['ldbw'] = tmp['t_dat_tmp'] - pd.TimedeltaIndex(tmp['dow'] - 1, unit='D')\ntmp.loc[tmp['dow'] >=2 , 'ldbw'] = tmp.loc[tmp['dow'] >=2 , 'ldbw'] + pd.TimedeltaIndex(np.ones(len(tmp.loc[tmp['dow'] >=2])) * 7, unit='D')\nt['ldbw'] = tmp['ldbw'].values\ndel tmp","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:33:29.921313Z","iopub.execute_input":"2023-02-21T04:33:29.922209Z","iopub.status.idle":"2023-02-21T04:33:37.946516Z","shell.execute_reply.started":"2023-02-21T04:33:29.922170Z","shell.execute_reply":"2023-02-21T04:33:37.945326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weekly_sales = t[['ldbw', 'article_id', 't_dat_tmp']].groupby(['ldbw', 'article_id']).count().reset_index()\nweekly_sales = weekly_sales.rename(columns={'t_dat_tmp': 'count'})\n\nt = t.merge(weekly_sales, on=['ldbw', 'article_id'], how = 'left')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:33:37.948190Z","iopub.execute_input":"2023-02-21T04:33:37.948582Z","iopub.status.idle":"2023-02-21T04:33:38.126617Z","shell.execute_reply.started":"2023-02-21T04:33:37.948543Z","shell.execute_reply":"2023-02-21T04:33:38.125588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:33:38.128034Z","iopub.execute_input":"2023-02-21T04:33:38.128838Z","iopub.status.idle":"2023-02-21T04:33:38.170881Z","shell.execute_reply.started":"2023-02-21T04:33:38.128799Z","shell.execute_reply":"2023-02-21T04:33:38.169778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get item pairs","metadata":{}},{"cell_type":"code","source":"def cudf_groupby_head(df, groupby, head_count):\n    df = df.to_pandas()\n\n    head_df = df.groupby(groupby).head(head_count)\n\n    head_df = cudf.DataFrame(head_df)\n\n    return head_df\n\n\ndef create_pairs(transactions_df, week_number, pairs_per_item, verbose=True):\n    # get the dfs\n    working_t_df = transactions_df[[\"customer_id\", \"article_id\", \"week_number\"]].copy()\n\n    # we'll look for pairs in last week only\n    working_t_df = working_t_df.query(f\"week_number <= {week_number}\").copy()\n    pairs_t_df = working_t_df.query(f\"week_number == {week_number}\").copy()\n    pairs_t_df.columns = [\"customer_id\", \"pair_article_id\", \"week_number\"]\n\n    # drop week number, and drop duplicates\n    del working_t_df[\"week_number\"], pairs_t_df[\"week_number\"]\n    working_t_df = working_t_df.drop_duplicates()\n    pairs_t_df = pairs_t_df.drop_duplicates()\n\n    # setup the loop\n    unique_articles = working_t_df[\"article_id\"].unique()\n    batch_size = 5000\n    batch_pairs_dfs = []\n\n    # start the loop\n    for i in range(0, len(unique_articles), batch_size):\n        if verbose:\n            print(f\"processing article #{i:,} to #{i+batch_size:,}\")\n\n        # take batch of articles/transactions\n        batch_articles = unique_articles[i : i + batch_size]\n        batch_t = working_t_df[working_t_df[\"article_id\"].isin(batch_articles)]\n\n        # get total # of customers who bought each article (for later statistics)\n        all_cust_counts = batch_t.groupby(\"article_id\")[\"customer_id\"].nunique()\n        all_cust_counts = all_cust_counts.reset_index()\n        all_cust_counts.columns = [\"article_id\", \"all_customer_counts\"]\n        all_cust_counts[\"all_customer_counts\"] -= 1  # not him himself\n\n        # get all pairs for those articles (other articles those customers bought)\n        batch_pairs_df = batch_t.merge(pairs_t_df, on=\"customer_id\")\n\n        # delete same-article pairs\n        same_article_row_idxs = batch_pairs_df.query(\n            \"article_id==pair_article_id\"\n        ).index\n        batch_pairs_df = batch_pairs_df.drop(same_article_row_idxs)\n\n        # delete single customer articles\n        c1s = (\n            batch_pairs_df.groupby(\"article_id\")[[\"customer_id\"]]\n            .nunique()\n            .query(\"customer_id==1\")\n            .index\n        )\n        single_customer_row_idxs = batch_pairs_df[\n            batch_pairs_df[\"article_id\"].isin(c1s)\n        ].index\n        batch_pairs_df = batch_pairs_df.drop(single_customer_row_idxs)\n\n        # get sorted counts of article-pair occurences\n        batch_pairs_df = batch_pairs_df.groupby([\"article_id\", \"pair_article_id\"])[\n            [\"customer_id\"]\n        ].count()\n        batch_pairs_df.columns = [\"customer_count\"]\n        batch_pairs_df = batch_pairs_df.reset_index()\n        batch_pairs_df = batch_pairs_df.sort_values(\n            [\"article_id\", \"customer_count\"], ascending=False\n        )\n\n        # get top x pairs for each article\n        batch_pairs_df = cudf_groupby_head(batch_pairs_df, \"article_id\", pairs_per_item)\n\n        # calculate percentage statistic\n        batch_pairs_df = batch_pairs_df.merge(all_cust_counts, on=\"article_id\")\n        batch_pairs_df[\"percent_customers\"] = (\n            batch_pairs_df[\"customer_count\"] / batch_pairs_df[\"all_customer_counts\"]\n        )\n        del batch_pairs_df[\"all_customer_counts\"]\n\n        batch_pairs_dfs.append(batch_pairs_df)\n\n    all_article_pairs_df = cudf.concat(batch_pairs_dfs)\n\n    return all_article_pairs_df","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:33:38.172538Z","iopub.execute_input":"2023-02-21T04:33:38.172931Z","iopub.status.idle":"2023-02-21T04:33:38.186939Z","shell.execute_reply.started":"2023-02-21T04:33:38.172895Z","shell.execute_reply":"2023-02-21T04:33:38.185878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npairs_per_item = 5\n\nweek_number_pairs = {}\nfor week_number in [96, 97, 98, 99, 100, 101, 102, 103, 104]:\n    print(f\"Creating pairs for week number {week_number}\")\n    week_number_pairs[week_number] = create_pairs(\n        t, week_number, pairs_per_item, verbose=False\n    )","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:33:38.188443Z","iopub.execute_input":"2023-02-21T04:33:38.189098Z","iopub.status.idle":"2023-02-21T04:34:45.160541Z","shell.execute_reply.started":"2023-02-21T04:33:38.189062Z","shell.execute_reply":"2023-02-21T04:34:45.158576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main retrieval/features function!","metadata":{}},{"cell_type":"code","source":"def ground_truth(transactions_df):\n    customer_trans = transactions_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n    customer_trans = customer_trans.groupby(\"customer_id\")[[\"article_id\"]].agg(list)\n    customer_trans.columns = [\"prediction\"]\n\n    gt = customer_trans.reset_index()\n\n    return gt\n\n\ndef feature_label_split(\n    transactions_df: cudf.DataFrame,\n    label_week_number: int,\n    feature_week_length=None,\n):\n    \"\"\"\n    split transaction_df into train/validation\n    use week numbers created already using fe.day_week_numbers() function\n    \"\"\"\n\n    # use week numbers to filter/create train/validation dfs\n    label_df = transactions_df.query(f\"week_number=={label_week_number}\").copy()\n    features_df = transactions_df.query(f\"week_number < {label_week_number}\").copy()\n    if feature_week_length is not None:\n        features_df = features_df.query(\n            f\"week_number >= {label_week_number - feature_week_length}\"\n        ).copy()\n\n    return features_df, label_df\n\n\ndef report_candidates(candidates: List[str], ground_truth_candidates: List[str]):\n    \"\"\"\n    Calculate recall and candidates factor (num(candidates) / num(ground_truth_candidates))\n    \"\"\"\n\n    num_candidates = len(candidates)\n    num_ground_truth = len(ground_truth_candidates)\n    all_candidates = cudf.concat(\n        [ground_truth_candidates, candidates]\n    ).drop_duplicates()\n    num_true_candidates = num_candidates + num_ground_truth - len(all_candidates)\n    recall = num_true_candidates / num_ground_truth\n    precision = num_true_candidates / num_candidates\n\n    print(\n        f\"candidates recall: {recall:.2%} ({num_true_candidates:,}/{num_ground_truth:,})\"\n    )\n    print(\n        f\"candidates precision: {precision:.2%} ({num_true_candidates:,}/{num_candidates:,})\"\n    )\n\n    return recall, precision\n\n\ndef comp_average_precision(\n    data_true: cudf.Series,\n    data_predicted: cudf.Series,\n) -> float:\n    \"\"\"\n    :param data_true: items that were actually purchased by user\n    :param data_predicted: items we recommended to user\n    \"\"\"\n    data_true = data_true.to_pandas().apply(list)\n    data_predicted = data_predicted.to_pandas().apply(list)\n\n    # for this competition, we don't score ones without any purchases\n    data_true = data_true[data_true.notna()]\n    data_true = data_true[data_true.apply(len) > 0]\n\n    if len(data_true) == 0:\n        raise ValueError(\"data_true is empty\")\n\n    # convert to df so we can use `apply`\n    eval_df = pd.DataFrame({\"true\": data_true})\n\n    eval_df[\"predicted\"] = data_predicted  # this way only get ones in data_true\n\n    # replace na predictions with empty list\n    eval_df[\"predicted\"] = eval_df[\"predicted\"].apply(\n        lambda x: x if isinstance(x, list) else []\n    )\n\n    # getting the counts of true/predicted\n    eval_df[\"n_items_true\"] = eval_df[\"true\"].apply(len)\n    eval_df[\"n_items_predicted\"] = eval_df[\"predicted\"].apply(lambda x: min(len(x), 12))\n\n    # ignore zero predicted (zero true is not even included...)\n    non_zero_filter = eval_df[\"n_items_predicted\"] > 0\n    eval_df = eval_df[non_zero_filter].copy()\n\n    def row_precision(items_true, items_predicted, n_items_predicted, n_items_true):\n        n_correct_items = 0\n        precision = 0.0\n\n        for item_idx in range(n_items_predicted):\n            if items_predicted[item_idx] in items_true:\n                n_correct_items += 1\n                precision += n_correct_items / (item_idx + 1)\n\n        return precision / min(n_items_true, 12)\n\n    eval_df[\"row_precision\"] = eval_df.apply(\n        lambda x: row_precision(\n            x[\"true\"], x[\"predicted\"], x[\"n_items_predicted\"], x[\"n_items_true\"]\n        ),\n        axis=1,\n    )\n\n    return eval_df[\"row_precision\"].sum() / len(data_true)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.162396Z","iopub.execute_input":"2023-02-21T04:34:45.162809Z","iopub.status.idle":"2023-02-21T04:34:45.183537Z","shell.execute_reply.started":"2023-02-21T04:34:45.162769Z","shell.execute_reply":"2023-02-21T04:34:45.182427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cudf_groupby_head(df, groupby, head_count):\n    df = df.to_pandas()\n\n    head_df = df.groupby(groupby).head(head_count)\n\n    head_df = cudf.DataFrame(head_df)\n\n    return head_df\n\n\ndef create_recent_customer_candidates(\n    transactions_df, recent_customer_weeks, customers=None\n):\n    if customers is not None:\n        transactions_df = transactions_df[\n            transactions_df[\"customer_id\"].isin(customers)\n        ]\n\n    last_week_number = transactions_df[\"week_number\"].max()\n\n    recent_customer_df = (\n        transactions_df.groupby([\"customer_id\", \"article_id\"])\n        .agg(\n            {\n                \"week_number\": \"max\",\n                \"t_dat\": \"max\",\n                \"price\": \"count\",\n            }\n        )\n        .rename(\n            columns={\n                \"week_number\": \"ca_last_purchase_week\",\n                \"t_dat\": \"ca_last_purchase_date\",\n                \"price\": \"ca_purchase_count\",\n            }\n        )\n        .sort_values(\"ca_purchase_count\", ascending=False)\n    )\n\n    features = ([\"customer_id\", \"article_id\"], recent_customer_df)\n    recent_customer_cand = (\n        recent_customer_df.query(\n            f\"ca_last_purchase_week >= {last_week_number - recent_customer_weeks + 1}\"\n        )\n        .reset_index()[[\"customer_id\", \"article_id\"]]\n        .drop_duplicates()\n    )\n\n    return recent_customer_cand, features\n\n\ndef create_repurchased_candidates(\n    transactions_df,\n    customers\n):  \n    \n    if customers is not None:\n        transactions_df = transactions_df[\n            transactions_df[\"customer_id\"].isin(customers)\n        ]\n    \n    weekly_sales_tmp = weekly_sales.reset_index().set_index('article_id').copy()\n    last_week_number = transactions_df[\"week_number\"].max()\n    \n    df = transactions_df.copy()\n    last_ts = df['t_dat_tmp'].max()\n    df = df.merge(\n    weekly_sales_tmp.loc[weekly_sales_tmp['ldbw']==last_ts, ['count']],\n    on='article_id', suffixes=(\"\", \"_targ\"))\n    del weekly_sales_tmp\n    \n    df['count_targ'].fillna(0, inplace=True)\n    df['quotient'] = df['count_targ'] / df['count']\n    \n    target_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\n    general_pred = target_sales.nlargest(12).index.to_pandas().tolist()\n    del target_sales\n     \n    cus_list = transactions_df['customer_id'].unique().to_arrow().to_pylist()\n    cans = itertools.product(cus_list,general_pred)\n    df_cans = cudf.from_pandas(pd.DataFrame(cans,columns=[\"customer_id\",\"article_id\"]))\n    del cans, general_pred, cus_list\n    \n    tmp = df.copy().to_pandas()\n    del df\n    tmp['x'] = ((last_ts - tmp['t_dat_tmp']) / np.timedelta64(1, 'D')).astype(int)\n    # convert 1\n    tmp = cudf.from_pandas(tmp)\n    tmp['dummy_1'] = 1 \n    tmp['x'] = tmp[[\"x\", \"dummy_1\"]].max(axis=1)\n\n    a, b, c, d = 2.5e4, 1.5e5, 2e-1, 1e3\n    tmp['y'] = a / np.sqrt(tmp['x']) + b * np.exp(-c*tmp['x']) - d\n\n    tmp['dummy_0'] = 0 \n    tmp['y'] = tmp[[\"y\", \"dummy_0\"]].max(axis=1)\n    tmp['value'] = tmp['quotient'] * tmp['y'] \n    \n    out_feature = tmp.drop_duplicates(\n        [\"customer_id\", \"article_id\"]\n    )[[\"customer_id\", \"article_id\", \"quotient\", \"y\"]]\n\n    features = ([\"customer_id\", \"article_id\"], out_feature)\n    \n    tmp = tmp.groupby(['customer_id', 'article_id']).agg({'value': 'sum'})\n    tmp = tmp.reset_index()\n\n    tmp = tmp.loc[tmp['value'] > 100]\n    # convert 2\n    tmp = tmp.to_pandas()\n    tmp['rank'] = tmp.groupby(\"customer_id\")[\"value\"].rank(\"dense\", ascending=False)\n    tmp = tmp.loc[tmp['rank'] <= 12]\n    purchase_df = tmp.sort_values(['customer_id', 'value'], ascending = False).reset_index(drop = True)\n    purchase_df = purchase_df[['customer_id', 'article_id']]\n    purchase_df = cudf.from_pandas(purchase_df)\n    purchase_df = cudf.concat([purchase_df, df_cans])\n    \n    del tmp, df_cans\n    \n    return purchase_df, features\n\ndef create_last_customer_weeks_and_pairs(\n    transactions_df, article_pairs_df, num_weeks, num_pair_weeks, customers\n):\n    clw_df = transactions_df[[\"customer_id\", \"article_id\", \"t_dat\"]].copy()\n    if customers is not None:\n        clw_df = clw_df[clw_df[\"customer_id\"].isin(customers)]\n\n    # only transactions in \"x\" weeks before last customer purchase\n    last_customer_purchase_dat = clw_df.groupby(\"customer_id\")[\"t_dat\"].max()\n    clw_df[\"max_cust_dat\"] = clw_df[\"customer_id\"].map(last_customer_purchase_dat)\n    clw_df[\"sample\"] = 1\n\n    clw_df = (\n        clw_df.groupby([\"customer_id\", \"article_id\"])\n        .agg(\n            {\n                \"max_cust_dat\": \"max\",\n                \"sample\": \"count\",\n                \"t_dat\": \"max\",\n            }\n        )\n        .rename(\n            columns={\n                \"max_cust_dat\": \"last_c_purchase_date\",\n                \"sample\": \"ca_count\",\n                \"t_dat\": \"last_ca_purchase_date\",\n            }\n        )\n        .reset_index()\n    )\n    clw_df[\"last_ca_purchase_diff\"] = (\n        clw_df[\"last_c_purchase_date\"] - clw_df[\"last_ca_purchase_date\"]\n    )\n\n    clw_pairs_df = clw_df.query(\n        f\"last_ca_purchase_diff <= {num_pair_weeks * 7 - 1}\"\n    ).copy()\n    clw_df = clw_df.query(f\"last_ca_purchase_diff <= {num_weeks * 7 - 1}\").copy()\n\n    del last_customer_purchase_dat\n\n    # merge with pairs, and get max of:\n    #  - sources' last week(s) purchase count\n    #  - count and percent of customer pairs (see generating code for details)\n    clw_pairs_df = clw_pairs_df.merge(article_pairs_df, on=\"article_id\")\n\n    clw_pairs_df = (\n        clw_pairs_df.groupby([\"customer_id\", \"pair_article_id\"])[\n            [\n                \"ca_count\",\n                \"last_ca_purchase_date\",\n                \"last_ca_purchase_diff\",\n                \"customer_count\",\n                \"percent_customers\",\n            ]\n        ]\n        .max()\n        .reset_index()\n    )\n    clw_pairs_df.columns = [\n        \"customer_id\",\n        \"article_id\",\n        \"pair_ca_count\",\n        \"pair_last_ca_purchase_date\",\n        \"pair_last_ca_purchase_diff\",\n        \"pair_customer_count\",\n        \"pair_percent_customers\",\n    ]\n    clw_pairs_df = clw_pairs_df.query(\"pair_customer_count > 2\").copy()\n\n    cust_last_week_cand = clw_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n    cust_last_week_pair_cand = clw_pairs_df[\n        [\"customer_id\", \"article_id\"]\n    ].drop_duplicates()\n\n    clw_df = clw_df.set_index([\"customer_id\", \"article_id\"])[\n        [\"ca_count\", \"last_ca_purchase_date\", \"last_ca_purchase_diff\"]\n    ].copy()\n    features = ([\"customer_id\", \"article_id\"], clw_df)\n\n    clw_pairs_df = clw_pairs_df.set_index([\"customer_id\", \"article_id\"])[\n        [\n            \"pair_ca_count\",\n            \"pair_last_ca_purchase_date\",\n            \"pair_last_ca_purchase_diff\",\n            \"pair_customer_count\",\n            \"pair_percent_customers\",\n        ]\n    ].copy()\n    pair_features = ([\"customer_id\", \"article_id\"], clw_pairs_df)\n\n    return cust_last_week_cand, cust_last_week_pair_cand, features, pair_features\n\n\ndef create_popular_article_cand(\n    transactions_df,\n    customers_df,\n    articles_df,\n    num_weeks,\n    hier_col,\n    num_candidates,\n    num_articles=12,\n    customers=None,\n):\n    ###########################################\n    # first get general popular candidates\n    ###########################################\n    last_week_number = transactions_df[\"week_number\"].max()\n\n    # baseline\n    article_purchases_df = (\n        transactions_df.query(f\"week_number >= {last_week_number - num_weeks + 1}\")\n        .groupby(\"article_id\")[\"customer_id\"]\n        .count()\n        .sort_values(ascending=False)\n    )\n    article_purchases_df = article_purchases_df.reset_index()\n    article_purchases_df.columns = [\"article_id\", \"counts\"]\n    popular_articles_df = article_purchases_df[:num_candidates].copy()\n    popular_articles_df[\"join_col\"] = 1\n\n    # from here on, only care about relevant customers\n    if customers is not None:\n        transactions_df = transactions_df[\n            transactions_df[\"customer_id\"].isin(customers)\n        ]\n        customers_df = customers_df[customers_df[\"customer_id\"].isin(customers)]\n\n    popular_articles_cand = cudf.DataFrame(\n        {\"customer_id\": customers_df[\"customer_id\"], \"join_col\": 1}\n    )\n    popular_articles_cand = popular_articles_cand.merge(\n        popular_articles_df, on=\"join_col\"\n    )\n    del popular_articles_cand[\"join_col\"]\n\n    ###################################################\n    # now let's limit it by cust/hierarchy information\n    ###################################################\n    sample_col = \"t_dat\"\n\n    # add hierarchy column to transactions\n    transactions_df[hier_col] = transactions_df[\"article_id\"].map(\n        articles_df.set_index(\"article_id\")[hier_col]\n    )\n    # get customer/hierarchy statistics\n    cust_hier = (\n        transactions_df.groupby([\"customer_id\", hier_col])[sample_col]\n        .count()\n        .reset_index()\n    )\n    cust_hier.columns = list(cust_hier.columns)[:-1] + [\"cust_hier_counts\"]\n    cust_hier = cust_hier.sort_values(\n        [\"customer_id\", \"cust_hier_counts\"], ascending=False\n    )\n    cust_hier[\"total_counts\"] = cust_hier[\"customer_id\"].map(\n        transactions_df.groupby(\"customer_id\")[sample_col].count()\n    )\n    cust_hier[\"cust_hier_portion\"] = (\n        cust_hier[\"cust_hier_counts\"] / cust_hier[\"total_counts\"]\n    )\n    cust_hier = cust_hier[[\"customer_id\", hier_col, \"cust_hier_portion\"]].copy()\n\n    # add customer/hierarchy statistics to candidates\n    popular_articles_cand[hier_col] = popular_articles_cand[\"article_id\"].map(\n        articles_df.set_index(\"article_id\")[hier_col]\n    )\n    popular_articles_cand = popular_articles_cand.merge(\n        cust_hier, on=[\"customer_id\", hier_col], how=\"left\"\n    )\n    popular_articles_cand[\"cust_hier_portion\"] = popular_articles_cand[\n        \"cust_hier_portion\"\n    ].fillna(-1)\n\n    del popular_articles_cand[hier_col]\n\n    # take top based on customer/hierarchy statistics\n    popular_articles_cand = popular_articles_cand.sort_values(\n        [\"customer_id\", \"cust_hier_portion\", \"counts\"], ascending=False\n    )\n    popular_articles_cand = popular_articles_cand[[\"customer_id\", \"article_id\"]].copy()\n    popular_articles_cand = cudf_groupby_head(\n        popular_articles_cand, \"customer_id\", num_articles\n    )\n    popular_articles_cand = popular_articles_cand.sort_values(\n        [\"customer_id\", \"article_id\"]\n    )\n    popular_articles_cand = popular_articles_cand.reset_index(drop=True)\n\n    # and save the article purchase statistics\n    article_purchases_df = article_purchases_df[[\"article_id\", \"counts\"]]\n    article_purchases_df.columns = [\"article_id\", \"recent_popularity_counts\"]\n    article_purchase_features = (\n        [\"article_id\"],\n        article_purchases_df.set_index(\"article_id\"),\n    )\n\n    return popular_articles_cand, article_purchase_features\n\n\ndef create_age_bucket_candidates(\n    transactions_df, customers_df, age_buckets, customers=None, articles=12\n):\n    # get transactions we're working with\n    working_t_df = transactions_df.copy()\n    working_t_df = working_t_df.drop_duplicates(\n        [\"customer_id\", \"article_id\", \"week_number\"]\n    )\n\n    # create the buckets\n    buckets_df = (\n        customers_df[[\"customer_id\"]].drop_duplicates().set_index(\"customer_id\")\n    )\n    buckets_df[\"age\"] = customers_df.set_index(\"customer_id\").age\n    buckets_df[\"age_bucket\"] = pd.qcut(\n        buckets_df[\"age\"].to_pandas(), age_buckets\n    ).cat.codes\n\n    # choose bucket\n    selected_buckets = [\"age_bucket\"]\n\n    # add the buckets to the transactions\n    working_t_df = working_t_df.merge(\n        buckets_df[selected_buckets].reset_index(), on=\"customer_id\"\n    )\n\n    # get the popularity\n    last_week = working_t_df[\"week_number\"].max()\n    pi_df = (\n        working_t_df.query(f\"week_number=={last_week}\")\n        .groupby(selected_buckets + [\"article_id\"])[\"t_dat\"]\n        .count()\n        .reset_index()\n        .sort_values(selected_buckets + [\"t_dat\"], ascending=False)\n    )\n    pi_df = cudf_groupby_head(pi_df, selected_buckets, articles)\n\n    # candidates - merge customer with their bucket\n    can_df = buckets_df.reset_index()[[\"customer_id\"] + selected_buckets].merge(\n        pi_df, on=selected_buckets\n    )\n    can_df.columns = [\"customer_id\", \"age_bucket\", \"article_id\", \"article_bucket_count\"]\n\n    # features dfs\n    buckets_df = buckets_df[[\"age_bucket\"]].copy()\n    bucket_counts_df = can_df[\n        [\"customer_id\", \"article_id\", \"article_bucket_count\"]\n    ].copy()\n    bucket_counts_df = bucket_counts_df.set_index([\"customer_id\", \"article_id\"])\n\n    # candidates_df\n    can_df = can_df[[\"customer_id\", \"article_id\"]]\n    if customers is not None:\n        can_df = can_df[can_df[\"customer_id\"].isin(customers)]\n\n    return (\n        can_df,\n        ([\"customer_id\"], buckets_df),\n        ([\"customer_id\", \"article_id\"], bucket_counts_df),\n    )\n\n\ndef add_features_to_candidates(candidates_df, features, customers_df, articles_df):\n    \"\"\"\n    adds fields needed to merge in features\n    and merges features in\n    \"\"\"\n    for features_key in features:\n        col_names, feature_df = features[features_key]\n\n        # add the key to our df so we can merge the features in\n        to_delete = []\n        for col_name in col_names:\n            if col_name not in candidates_df:\n                if col_name in customers_df:\n                    col_name_dict = customers_df.set_index(\"customer_id\")[col_name]\n                    candidates_df[col_name] = candidates_df[\"customer_id\"].map(\n                        col_name_dict\n                    )\n                    to_delete.append(col_name)\n                elif col_name in articles_df:\n                    col_name_dict = articles_df.set_index(\"article_id\")[col_name]\n                    candidates_df[col_name] = candidates_df[\"article_id\"].map(\n                        col_name_dict\n                    )\n                    to_delete.append(col_name)\n\n        # now we can add the features\n        candidates_df = candidates_df.merge(feature_df, how=\"left\", on=col_names)\n\n        for col_name in to_delete:\n            del candidates_df[col_name]\n\n    return candidates_df\n\n\ndef filter_candidates(candidates, transactions_df, **kwargs):\n    recent_art_weeks = kwargs[\"filter_recent_art_weeks\"]\n    recent_articles = transactions_df.query(\n        f\"week_number >= {kwargs['label_week'] - recent_art_weeks}\"\n    )[\"article_id\"]\n\n    num_articles = kwargs.get(\"filter_num_articles\", None)\n    if num_articles is None:\n        recent_articles = recent_articles.drop_duplicates()\n    else:\n        recent_item_counts = recent_articles.value_counts()\n        most_popular_items = recent_item_counts[:num_articles].index\n        most_popular_items = most_popular_items.to_pandas().to_list()\n        recent_articles = most_popular_items\n\n    candidates = candidates[candidates[\"article_id\"].isin(recent_articles)].copy()\n\n    return candidates","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.188232Z","iopub.execute_input":"2023-02-21T04:34:45.188578Z","iopub.status.idle":"2023-02-21T04:34:45.242238Z","shell.execute_reply.started":"2023-02-21T04:34:45.188546Z","shell.execute_reply":"2023-02-21T04:34:45.241110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_candidates_with_features_df(t, c, a, customer_batch=None, **kwargs):\n    # splitting cv\n    features_df, label_df = feature_label_split(\n        t, kwargs[\"label_week\"], kwargs[\"feature_periods\"]\n    )\n    \n    # converting relative day_number\n    features_df[\"t_dat\"] = how_many_ago(features_df[\"t_dat\"])\n    features_df[\"week_number\"] = how_many_ago(features_df[\"week_number\"])\n    \n    # pull out the cv week\n    article_pairs_df = week_number_pairs[kwargs[\"label_week\"]-1]\n    \n    # check if we can limit customers\n    if len(label_df) > 0:\n        customers = label_df[\"customer_id\"].unique()\n    elif customer_batch is not None:\n        customers = customer_batch\n    else:\n        customers = None\n    \n    ############################################\n    # creating candidates (and adding features)\n    ###########################################\n    \n    features_db = shelve.open(\"features_db\") \n    \n    # creating candidate (and saving features created)\n    repurchased_article_cand, features_db[\"rebuy_rate\"] = (\n        create_repurchased_candidates(\n            features_df,\n            customers=customers,\n        )\n    )\n    recent_customer_cand, features_db[\"customer_article\"] = (\n        create_recent_customer_candidates(\n            features_df,\n            kwargs[\"ca_num_weeks\"],\n            customers=customers,\n        )\n    )\n    \n    (cust_last_week_cand,\n     cust_last_week_pair_cand,\n     features_db[\"clw\"],\n     features_db[\"clw_pairs\"]) = create_last_customer_weeks_and_pairs(\n        features_df,\n        article_pairs_df,\n        kwargs[\"clw_num_weeks\"],\n        kwargs[\"clw_num_pair_weeks\"],\n        customers=customers,\n    )\n    \n    _, features_db[\"popular_articles\"] = create_popular_article_cand(\n        features_df,\n        c,\n        a,\n        kwargs[\"pa_num_weeks\"],\n        kwargs[\"hier_col\"],\n        num_candidates=kwargs[\"num_recent_candidates\"],\n        num_articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    age_bucket_can, _, _ = create_age_bucket_candidates(\n        features_df,\n        c,\n        kwargs[\"num_age_buckets\"],\n        articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    \n    cand = [repurchased_article_cand, recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can]\n    cand = cudf.concat(cand).drop_duplicates()\n    cand = cand.sort_values([\"customer_id\", \"article_id\"]).reset_index(drop=True)\n    \n    del repurchased_article_cand, recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can, _\n    \n    cand = filter_candidates(cand, t, **kwargs)\n    \n    # creating other features\n    create_cust_hier_features(features_df, a, kwargs[\"hier_cols\"], features_db)\n    create_price_features(features_df, features_db)\n    create_cust_features(c, features_db)\n    create_article_cust_features(features_df, c, features_db)\n    create_lag_features(features_df, a, kwargs[\"lag_days\"], features_db)\n    create_rebuy_features(features_df, features_db)\n    create_cust_t_features(features_df, a, features_db)\n    create_art_t_features(features_df, features_db)\n    create_art_features(a, features_db)\n    \n    del features_df\n\n    # another filter at the end, for the ones that didn't get filtered earlier\n    if customers is not None:\n        cand = cand[cand[\"customer_id\"].isin(customers)]\n    \n    # report on recall/precision of candidates\n    if kwargs[\"cv\"]:\n        ground_truth_candidates = label_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n        report_candidates(cand, ground_truth_candidates)\n        del ground_truth_candidates        \n    \n    # adding features to candidates\n    cand_with_f_df = add_features_to_candidates(\n        cand, features_db, c, a\n    )\n    \n    # manually adding article features (couldn't use shelve for some reason)\n    for article_col in kwargs[\"article_columns\"]:\n        art_col_map = a.set_index(\"article_id\")[article_col]\n        cand_with_f_df[article_col] = cand_with_f_df[\"article_id\"].map(art_col_map)\n    \n    # limiting features\n    if kwargs[\"selected_features\"] is not None:\n        cand_with_f_df = cand_with_f_df[\n            [\"customer_id\", \"article_id\"] + kwargs[\"selected_features\"]\n        ]\n        \n    features_db.close()\n    os.remove(\"features_db.bak\"), os.remove(\"features_db.dir\"), os.remove(\"features_db.dat\")\n    \n    assert len(cand) == len(cand_with_f_df), \"seem to have duplicates in the feature dfs\"\n    del cand\n    gc.collect()\n    return cand_with_f_df, label_df","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.245095Z","iopub.execute_input":"2023-02-21T04:34:45.245976Z","iopub.status.idle":"2023-02-21T04:34:45.262409Z","shell.execute_reply.started":"2023-02-21T04:34:45.245934Z","shell.execute_reply":"2023-02-21T04:34:45.261431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model #","metadata":{}},{"cell_type":"code","source":"def get_group_lengths(df):\n    \"\"\"\n    Get group_lengths to pass to LGBMRanker\n    \"\"\"\n    df = df.to_pandas()\n\n    return list(df.groupby(\"customer_id\")[\"article_id\"].count())\n\n\ndef cudf_groupby_head(df, groupby, head_count):\n    df = df.to_pandas()\n\n    head_df = df.groupby(groupby).head(head_count)\n\n    head_df = cudf.DataFrame(head_df)\n\n    return head_df\n\n\ndef create_predictions(ids_df, preds):\n    ids_df[\"pred\"] = preds\n    ids_df = ids_df.sort_values([\"customer_id\", \"pred\"], ascending=False)\n    ids_df = cudf_groupby_head(ids_df, \"customer_id\", 12)\n    predictions = ids_df.groupby(\"customer_id\")[\"article_id\"].agg(list)\n\n    return predictions\n\n\ndef pred_in_batches(model, features_df, batch_size=1000000):\n    preds = []\n    num_batches = int(len(features_df) / 1000000) + 1\n    for batch in range(num_batches):\n        batch_df = features_df.iloc[batch * batch_size : (batch + 1) * batch_size, :]\n        preds.append(model.predict(batch_df))\n\n    return np.concatenate(preds)\n\n\ndef prep_cudf_to_pandas(cudf_df, inplace=False):\n    \"\"\"\n    prepared a cudf df for efficient conversion to pandas\n    by converting all int columns with null values to float32\n    (otherwise they'd become float 64 in pandas)\n\n    if inplace=True, will convert inplace and return None\n    if inplace=False, will create copy, convert and return\n\n    (meant for use on cudf df, but won't fail on pandas df)\n    \"\"\"\n    if not inplace:\n        working_df = cudf_df.copy()\n    else:\n        working_df = cudf_df\n\n    for col_name in working_df:\n        is_int_type = str(working_df[col_name].dtype)[:3] == \"int\"\n        has_nulls = working_df[col_name].isna().mean() > 0\n        if is_int_type and has_nulls:\n            working_df[col_name] = working_df[col_name].astype(\"float32\")\n\n    if not inplace:\n        return working_df\n    else:\n        return None\n\n\ndef prepare_modeling_dfs(t, c, a, cand_features_func, **params):\n    # \"cf\" means -> candidates_with_features\n    cf_df, label_df = cand_features_func(t, c, a, **params)\n\n    # adding y column\n    label_df = label_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n    label_df[\"match\"] = 1\n    cf_df = cf_df.merge(label_df, how=\"left\", on=[\"customer_id\", \"article_id\"])\n    cf_df[\"match\"] = cf_df[\"match\"].fillna(0).astype(\"int8\")\n    cf_df = cf_df.sample(frac=1, random_state=42).reset_index(drop=True)\n    del label_df[\"match\"]\n\n    # only customer with some positives ones\n    customers_with_positives = cf_df.query(\"match==1\")[\"customer_id\"].unique()\n    cf_df = cf_df[cf_df[\"customer_id\"].isin(customers_with_positives)]\n    cf_df = cf_df.sort_values(\"customer_id\").reset_index(drop=True)\n\n    # get group lengths\n    group_lengths = get_group_lengths(cf_df)\n\n    # split out ids\n    ids_df = cf_df[[\"customer_id\", \"article_id\"]].copy()\n    del cf_df[\"customer_id\"], cf_df[\"article_id\"]\n\n    # split out y\n    y = cf_df[[\"match\"]].copy()\n    del cf_df[\"match\"]\n\n    # rename\n    X = cf_df\n    del cf_df\n\n    ground_truth_df = label_df\n    del label_df\n\n    # prep for pandas\n    prep_cudf_to_pandas(X, inplace=True)\n    prep_cudf_to_pandas(y, inplace=True)\n    X = X.to_pandas()\n    y = y.to_pandas()\n\n    y = y[\"match\"]\n\n    return ids_df, X, group_lengths, y, ground_truth_df\n\n\ndef prepare_prediction_dfs(t, c, a, cand_features_func, customer_batch=None, **params):\n    # \"cf\" means -> candidates_with_features\n    cf_df, _ = cand_features_func(t, c, a, customer_batch=customer_batch, **params)\n\n    # split out ids\n    ids_df = cf_df[[\"customer_id\", \"article_id\"]].copy()\n    del cf_df[\"customer_id\"], cf_df[\"article_id\"]\n\n    # rename\n    X = cf_df\n    del cf_df\n\n    # prep for pandas\n    prep_cudf_to_pandas(X, inplace=True)\n    X = X.to_pandas()\n\n    return ids_df, X\n\n\ndef prepare_concat_train_modeling_dfs(t, c, a, cand_features_func, **params):\n    working_params = copy.deepcopy(params)\n\n    if params[\"num_concats\"] > 1:\n        empty_list = [None] * params[\"num_concats\"]\n        empty_lists = [empty_list.copy() for i in range(5)]\n\n        (\n            train_ids_df,\n            train_X,\n            train_group_lengths,\n            train_y,\n            train_truth_df,\n        ) = empty_lists\n\n        for i in range(params[\"num_concats\"]):\n            week_num = params[\"label_week\"] - (i + 1)\n            print(f\"preparing training modeling dfs for {week_num}...\")\n            working_params[\"label_week\"] = week_num\n            (\n                train_ids_df[i],\n                train_X[i],\n                train_group_lengths[i],\n                train_y[i],\n                train_truth_df[i],\n            ) = prepare_modeling_dfs(t, c, a, cand_features_func, **working_params)\n\n        print(\"concatenating all weeks together\")\n        train_ids_df = cudf.concat(train_ids_df)\n        train_group_lengths = sum(train_group_lengths, [])\n        train_truth_df = cudf.concat(train_truth_df)\n        train_X = pd.concat(train_X)\n        train_y = pd.concat(train_y)\n    else:\n        week_num = params[\"label_week\"] - 1\n        print(f\"preparing training modeling dfs for {week_num}...\")\n        working_params[\"label_week\"] = week_num\n        (\n            train_ids_df,\n            train_X,\n            train_group_lengths,\n            train_y,\n            train_truth_df,\n        ) = prepare_modeling_dfs(t, c, a, cand_features_func, **working_params)\n\n    return train_ids_df, train_X, train_group_lengths, train_y, train_truth_df\n\n\ndef prepare_train_eval_modeling_dfs(t, c, a, cand_features_func, **params):\n    (\n        train_ids_df,\n        train_X,\n        train_group_lengths,\n        train_y,\n        train_truth_df,\n    ) = prepare_concat_train_modeling_dfs(t, c, a, cand_features_func, **params)\n\n    print(\"preparing evaluation modeling dfs...\")\n    (\n        eval_ids_df,\n        eval_X,\n        eval_group_lengths,\n        eval_y,\n        eval_truth_df,\n    ) = prepare_modeling_dfs(t, c, a, cand_features_func, **params)\n\n    return (\n        train_ids_df,\n        train_X,\n        train_group_lengths,\n        train_y,\n        train_truth_df,\n        eval_ids_df,\n        eval_X,\n        eval_group_lengths,\n        eval_y,\n        eval_truth_df,\n    )\n\n\ndef full_cv_run(t, c, a, cand_features_func, score_func, **kwargs):\n    # load training data\n    train_eval_dfs = prepare_train_eval_modeling_dfs(\n        t, c, a, cand_features_func, **kwargs\n    )\n    (\n        train_ids_df,\n        train_X,\n        train_group_lengths,\n        train_y,\n        train_truth_df,\n        eval_ids_df,\n        eval_X,\n        eval_group_lengths,\n        eval_y,\n        eval_truth_df,\n    ) = train_eval_dfs\n\n    model = LGBMRanker(**(kwargs[\"lgbm_params\"]), seed=42)\n\n    eval_set = [(train_X, train_y), (eval_X, eval_y)]\n    eval_group = [train_group_lengths, eval_group_lengths]\n    eval_names = [\"train\", \"validation\"]\n\n    le_callback = lightgbm.log_evaluation(kwargs[\"log_evaluation\"])\n    es_callback = lightgbm.early_stopping(kwargs[\"early_stopping\"])\n\n    model.fit(\n        train_X,\n        train_y,\n        eval_set=eval_set,\n        eval_names=eval_names,\n        eval_group=eval_group,\n        eval_metric=\"MAP\",\n        eval_at=kwargs[\"eval_at\"],\n        callbacks=[le_callback, es_callback],\n        group=train_group_lengths,\n    )\n\n    if kwargs.get(\"save_model\", False):\n        with open(f\"model_{kwargs['label_week']}\", \"wb\") as f:\n            pkl.dump(model, f)\n\n    # train predictions and scores\n    train_pred = model.predict(train_X)\n    print(\"Train AUC {:.4f}\".format(roc_auc_score(train_y, train_pred)))\n    print(\"Train score: \", score_func(train_ids_df, train_pred, train_truth_df))\n\n    del train_X, train_y, train_group_lengths, train_truth_df, train_pred\n\n    # evaluation predictions and scores\n    eval_pred = pred_in_batches(model, eval_X)\n\n    print(\"Eval AUC {:.4f}\".format(roc_auc_score(eval_y, eval_pred)))\n    eval_score = score_func(eval_ids_df, eval_pred, eval_truth_df)\n    print(\"Eval score:\", eval_score)\n\n    # print feature importances\n    feature_importance_dict = dict(\n        zip(list(eval_X.columns), list(model.feature_importances_))\n    )\n    feature_importance_series = cudf.Series(feature_importance_dict).sort_values()\n    print(\"\\n\")\n    print(feature_importance_series)\n\n    # sorted(zip(clf.feature_importances_, X.columns), reverse=True)\n    feature_imp = pd.DataFrame(sorted(zip(model.feature_importances_,eval_X.columns)), columns=['Value','Feature'])\n\n    plt.figure(figsize=(20, 10))\n    sns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\", ascending=False))\n    plt.title('LightGBM Features (avg over folds)')\n    plt.tight_layout()\n    plt.show()\n\n    return eval_score\n\n\ndef run_all_cvs(\n    t, c, a, cand_features_func, score_func, cv_weeks=[102, 103, 104], **params\n):\n    cv_scores = []\n    total_duration = datetime.now() - datetime.now()\n\n    for cv_week in cv_weeks:\n        starting_time = datetime.now()\n\n        cv_params = copy.deepcopy(params)\n        cv_params.update({\"label_week\": cv_week})\n        cv_score = full_cv_run(t, c, a, cand_features_func, score_func, **cv_params)\n\n        cv_scores.append(cv_score)\n        duration = datetime.now() - starting_time\n        total_duration += duration\n        print(f\"Finished cv of week {cv_week} in {duration}. Score: {cv_score}\\n\")\n\n    average_scores = round(np.mean(cv_scores), 5)\n    print(\n        f\"Finished all {len(cv_weeks)} cvs in {total_duration}. \"\n        f\"Average cv score: {average_scores}\"\n    )\n\n    return average_scores\n\n\ndef full_sub_train_run(t, c, a, cand_features_func, score_func, **kwargs):\n    (\n        train_ids_df,\n        train_X,\n        train_group_lengths,\n        train_y,\n        train_truth_df,\n    ) = prepare_concat_train_modeling_dfs(t, c, a, cand_features_func, **kwargs)\n\n    model = LGBMRanker(**(kwargs[\"lgbm_params\"]), seed=42)\n\n    eval_set = [(train_X, train_y)]\n    eval_group = [train_group_lengths]\n    eval_names = [\"train\"]\n\n    le_callback = lightgbm.log_evaluation(1)\n\n    model.fit(\n        train_X,\n        train_y,\n        eval_set=eval_set,\n        eval_names=eval_names,\n        eval_group=eval_group,\n        eval_metric=\"MAP\",\n        eval_at=kwargs[\"eval_at\"],\n        callbacks=[le_callback],\n        group=train_group_lengths,\n    )\n\n    # train predictions and scores\n    train_pred = model.predict(train_X)\n    print(\"Train AUC {:.4f}\".format(roc_auc_score(train_y, train_pred)))\n    print(\"Train score: \", score_func(train_ids_df, train_pred, train_truth_df))\n\n    del train_X, train_y, train_group_lengths, train_truth_df, train_pred\n\n    if kwargs.get(\"save_model\", False):\n        with open(f\"model_{kwargs['label_week']}\", \"wb\") as f:\n            pkl.dump(model, f)\n\n\ndef full_sub_predict_run(t, c, a, cand_features_func, **kwargs):\n    customer_batches = []\n\n    num_customers = len(c)\n    first = num_customers // 3\n    second = num_customers * 2 // 3\n\n    customer_batches.append(c[:first][\"customer_id\"].to_pandas().to_list())\n    customer_batches.append(c[first:second][\"customer_id\"].to_pandas().to_list())\n    customer_batches.append(c[second:][\"customer_id\"].to_pandas().to_list())\n\n    \n    \n    batch_preds = []\n    for idx, customer_batch in enumerate(customer_batches):\n        print(\n            f\"generating candidates/features for batch #{idx+1} of {len(customer_batches)}\"\n        )\n        sub_ids_df, sub_X = prepare_prediction_dfs(\n            t, c, a, cand_features_func, customer_batch=customer_batch, **kwargs\n        )\n\n        print(\n            f\"candidate/features shape of batch: ({sub_X.shape[0]:,}, {sub_X.shape[1]})\",\n        )\n\n        prediction_models = kwargs.get(\"prediction_models\")\n        model_nums = len(prediction_models)\n\n        first_model_path = prediction_models[0]\n        with open(first_model_path, \"rb\") as f:\n            first_model = pkl.load(f)\n\n        print(f\"predicting with '{first_model_path}'\")\n        sub_pred = pred_in_batches(first_model, sub_X) / model_nums\n        del first_model\n\n        for model_path in prediction_models[1:]:\n            with open(model_path, \"rb\") as f:\n                model = pkl.load(f)\n                print(f\"predicting with '{model_path}'\")\n                sub_pred2 = pred_in_batches(model, sub_X)\n                del model\n                sub_pred += sub_pred2 / model_nums\n\n        batch_preds.append(create_predictions(sub_ids_df, sub_pred))\n\n        del sub_ids_df, sub_X, sub_pred\n\n    predictions = cudf.concat(batch_preds)\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.264252Z","iopub.execute_input":"2023-02-21T04:34:45.264656Z","iopub.status.idle":"2023-02-21T04:34:45.537040Z","shell.execute_reply.started":"2023-02-21T04:34:45.264603Z","shell.execute_reply":"2023-02-21T04:34:45.536046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_model_score(ids_df, preds, truth_df):\n    predictions = create_predictions(ids_df, preds)\n    true_labels = ground_truth(truth_df).set_index(\"customer_id\")[\"prediction\"]\n    score = round(comp_average_precision(true_labels, predictions),5)\n    \n    return score","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.538594Z","iopub.execute_input":"2023-02-21T04:34:45.539259Z","iopub.status.idle":"2023-02-21T04:34:45.546168Z","shell.execute_reply.started":"2023-02-21T04:34:45.539217Z","shell.execute_reply":"2023-02-21T04:34:45.544010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parameters","metadata":{}},{"cell_type":"code","source":"cv_params = {\n    \"cv\": True,\n    \"feature_periods\": 105,\n    \"label_week\": 104,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 36,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\"n_estimators\": 200, \"num_leaves\": 20},\n    \"log_evaluation\": 10,\n    \"early_stopping\": 20,\n    \"eval_at\": 12,\n    \"save_model\": True,\n    \"num_concats\": 5,\n}\nsub_params = {\n    \"cv\": False,\n    \"feature_periods\": 105,\n    \"label_week\": 105,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 60,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\n        \"n_estimators\": 150,\n        \"num_leaves\": 20,    \n    },\n    \"log_evaluation\": 10,\n    \"eval_at\": 12,\n    \"prediction_models\": [\"model_104\", \"model_105\"],\n    \"save_model\": True,\n    \"num_concats\": 5,\n}","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.547971Z","iopub.execute_input":"2023-02-21T04:34:45.548678Z","iopub.status.idle":"2023-02-21T04:34:45.559916Z","shell.execute_reply.started":"2023-02-21T04:34:45.548563Z","shell.execute_reply":"2023-02-21T04:34:45.558924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cand_features_func = create_candidates_with_features_df\nscoring_func = calculate_model_score","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.561282Z","iopub.execute_input":"2023-02-21T04:34:45.561793Z","iopub.status.idle":"2023-02-21T04:34:45.574004Z","shell.execute_reply.started":"2023-02-21T04:34:45.561742Z","shell.execute_reply":"2023-02-21T04:34:45.572922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncv_weeks = [104]\nresults = run_all_cvs(\n    t, c, a, cand_features_func, scoring_func, \n    cv_weeks=cv_weeks, **cv_params\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:34:45.575559Z","iopub.execute_input":"2023-02-21T04:34:45.576099Z","iopub.status.idle":"2023-02-21T04:42:30.256597Z","shell.execute_reply.started":"2023-02-21T04:34:45.576063Z","shell.execute_reply":"2023-02-21T04:42:30.255654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ngc.collect()\nfull_sub_train_run(t, c, a, cand_features_func, scoring_func, **sub_params)\npredictions = full_sub_predict_run(\n    t, c, a, cand_features_func, **sub_params\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:42:30.257661Z","iopub.execute_input":"2023-02-21T04:42:30.258836Z","iopub.status.idle":"2023-02-21T04:59:04.476841Z","shell.execute_reply.started":"2023-02-21T04:42:30.258796Z","shell.execute_reply":"2023-02-21T04:59:04.475690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_duplicates_list(orig_list: list) -> list:\n    \"\"\"\n    remove duplicates from python list, while retaining order\n    \"\"\"\n\n    unique_list = copy.deepcopy(orig_list)\n\n    for idx in range(len(unique_list), 1, -1):\n        if unique_list[idx - 1] in unique_list[: idx - 1]:\n            unique_list.pop(idx - 1)\n\n    return unique_list\n\n\ndef create_sub(\n    customer_ids: Union[List[str], cudf.Series],\n    predictions: Union[dict, cudf.Series],\n    index_to_id_dict_path: str = None,\n    default_predictions: List[str] = [],\n) -> pd.Series:\n    \"\"\"\n    given predictions for some customers,\n    generates a pd.Series for all customers, in the competition submission format.\n\n    predictions are limited to first 12\n    customer ids without predictions provided will be blank\n    customer ids with less that 12 predictions (or none provided)\n        will be augmented with default_predictions, if provided\n        default_predictions will be added in the order they are present in the list\n\n    args:\n    customer_ids: all the article ids we need to submit for\n    predictions: pd.Series where index=customer_id and values=list of predictions\n    default_predictions: predictions to augment customer_id predictions with, until 12\n    index_to_id_dict_path: mapping to get from number index to the customer_id for submission\n    \"\"\"\n\n    # can't support cudf for map\n    predictions = predictions.to_pandas().apply(list)\n    customer_ids = customer_ids.to_pandas()\n\n    # original predictions or empty list\n    sub = pd.DataFrame({\"customer_id\": customer_ids})\n    sub[\"prediction\"] = sub[\"customer_id\"].map(predictions)\n    sub[\"prediction\"] = sub[\"prediction\"].apply(\n        lambda x: x if isinstance(x, list) else []\n    )\n\n    sub[\"prediction\"] = sub[\"prediction\"].apply(lambda x: x + default_predictions)\n    sub[\"prediction\"] = sub[\"prediction\"].apply(remove_duplicates_list)\n    sub[\"prediction\"] = sub[\"prediction\"].apply(lambda x: x[:12])\n    sub[\"prediction\"] = sub[\"prediction\"].apply(\n        lambda x: \" \".join([\"0\" + str(article_id) for article_id in x])\n    )\n\n    if index_to_id_dict_path is not None:\n        index_to_id_dict = pkl.load(open(index_to_id_dict_path, \"rb\"))\n\n        index_to_id_dict = index_to_id_dict.to_pandas()\n\n        sub[\"customer_id\"] = sub[\"customer_id\"].map(index_to_id_dict)\n\n    return sub","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:59:04.478408Z","iopub.execute_input":"2023-02-21T04:59:04.479909Z","iopub.status.idle":"2023-02-21T04:59:04.492602Z","shell.execute_reply.started":"2023-02-21T04:59:04.479868Z","shell.execute_reply":"2023-02-21T04:59:04.491473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = create_sub(c[\"customer_id\"], predictions, index_to_id_dict_path)\nsub.to_csv('dev_submission.csv', index=False)\n\ndisplay(sub.head())\nprint(sub.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T04:59:04.494015Z","iopub.execute_input":"2023-02-21T04:59:04.494463Z","iopub.status.idle":"2023-02-21T04:59:04.660949Z","shell.execute_reply.started":"2023-02-21T04:59:04.494424Z","shell.execute_reply":"2023-02-21T04:59:04.659520Z"},"trusted":true},"execution_count":null,"outputs":[]}]}