{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Original Notebook from cdeotte can be found [here](https://www.kaggle.com/code/cdeotte/customers-who-bought-this-frequently-buy-this).\n\n### [Here's](https://www.kaggle.com/code/cdeotte/recommend-items-purchased-together-0-021/comments#1703595) where cdeotte gives the code for running it on the entire dataset.  \n### And [here's](https://www.kaggle.com/code/cdeotte/recommend-items-purchased-together-0-021/comments#1732941) where he mentions that it could probably be rewritten to be a lot quicker.    ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport cudf\nimport pickle as pkl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# load transactions\nt = cudf.read_csv(\n    \"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\n    ,usecols=[\"customer_id\", \"article_id\"]\n)\nt = t.drop_duplicates()\nt[\"article_id\"] = t[\"article_id\"].astype(\"int32\")\n\n# convert customer_id field in transactions\nc = cudf.read_csv(\n    \"../input/h-and-m-personalized-fashion-recommendations/customers.csv\",\n    usecols=[\"customer_id\"]\n)\nc_id_to_index = c.reset_index().set_index(\"customer_id\")[\"index\"]\nt[\"customer_id\"] = t[\"customer_id\"].map(c_id_to_index)\nt[\"customer_id\"] = t[\"customer_id\"].astype(\"int32\")\ndel c, c_id_to_index\n\n# create pair_transactions copy\npairs_t = t.copy()\npairs_t.columns = [\"customer_id\", \"pair_id\"]\n\n# unique articles\nunique_articles = t[\"article_id\"].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nbatch_size = 5000\n\nbatch_pairs_dfs = []\n\nfor i in range(0, len(unique_articles), batch_size):\n    print(f\"processing article #{i:,} to #{i+batch_size:,}\")\n\n    # take batch of articles\n    batch_articles = unique_articles[i:i+batch_size]\n\n    # get all pairs for those articles (other articles those customers bought)\n    batch_t = t[t[\"article_id\"].isin(batch_articles)]\n    batch_pairs_df = batch_t.merge(pairs_t, on=\"customer_id\")\n\n    # delete same-article pairs\n    same_article_row_idxs = batch_pairs_df.query(\"article_id==pair_id\").index\n    batch_pairs_df = batch_pairs_df.drop(same_article_row_idxs)\n    \n    # delete single customer articles\n    c1s = (\n        batch_pairs_df.groupby(\"article_id\")[[\"customer_id\"]].nunique()\n        .query(\"customer_id==1\").index\n    )\n    single_customer_row_idxs = batch_pairs_df[batch_pairs_df[\"article_id\"].isin(c1s)].index\n    batch_pairs_df = batch_pairs_df.drop(single_customer_row_idxs)\n    \n    # get sorted counts of article-pair occurences\n    batch_pairs_df = batch_pairs_df.groupby([\"article_id\", \"pair_id\"])[[\"customer_id\"]].count()\n    batch_pairs_df.columns = [\"pair_counts\"]\n    batch_pairs_df = batch_pairs_df.reset_index()\n    batch_pairs_df = batch_pairs_df.sort_values([\"article_id\", \"pair_counts\"], ascending=False)\n\n    # get top one for each article (need pandas)\n    batch_pairs_df = batch_pairs_df.to_pandas().groupby(\"article_id\").head(1)\n    # back to cudf\n    batch_pairs_df = cudf.DataFrame(batch_pairs_df.set_index(\"article_id\")[[\"pair_id\"]])\n    batch_pairs_dfs.append(batch_pairs_df)\n    \nall_article_pairs_df = cudf.concat(batch_pairs_dfs)\nprint(len(all_article_pairs_df))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\narticle_pairs_dict = all_article_pairs_df[\"pair_id\"].to_pandas().to_dict()\nwith open(\"article_pairs_dict.pkl\", \"wb\") as f:\n    pkl.dump(article_pairs_dict, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}