{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom datetime import timedelta\nfrom ast import literal_eval\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T23:10:35.243150Z","iopub.execute_input":"2022-07-26T23:10:35.243501Z","iopub.status.idle":"2022-07-26T23:10:35.249359Z","shell.execute_reply.started":"2022-07-26T23:10:35.243471Z","shell.execute_reply":"2022-07-26T23:10:35.248200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfp_train = pd.read_csv(\"/kaggle/input/what-card-should-i-put-next/train.csv\")\ndfp_test = pd.read_csv(\"/kaggle/input/what-card-should-i-put-next/test.csv\")\ndfp_cards = pd.read_csv(\"/kaggle/input/what-card-should-i-put-next/cards.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:35.391204Z","iopub.execute_input":"2022-07-26T23:10:35.391817Z","iopub.status.idle":"2022-07-26T23:10:37.921878Z","shell.execute_reply.started":"2022-07-26T23:10:35.391777Z","shell.execute_reply":"2022-07-26T23:10:37.921073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfp_train[\"update_date\"] = pd.to_datetime(dfp_train[\"update_date\"])\nprint(\"Count of rows (dfp_train):\", len(dfp_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:37.923521Z","iopub.execute_input":"2022-07-26T23:10:37.923985Z","iopub.status.idle":"2022-07-26T23:10:38.065154Z","shell.execute_reply.started":"2022-07-26T23:10:37.923956Z","shell.execute_reply":"2022-07-26T23:10:38.063987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"days = 7\ndfp_train_select = dfp_train[dfp_train[\"update_date\"] >= dfp_train[\"update_date\"].max() - timedelta(days = days)].copy()\ndfp_train_select[\"cards\"] = dfp_train_select[\"cards\"].apply(lambda cards: literal_eval(cards))\nprint(\"Count of rows (dfp_train_select):\", len(dfp_train_select))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:38.066629Z","iopub.execute_input":"2022-07-26T23:10:38.066999Z","iopub.status.idle":"2022-07-26T23:10:38.196776Z","shell.execute_reply.started":"2022-07-26T23:10:38.066966Z","shell.execute_reply":"2022-07-26T23:10:38.195610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfp_train_select_exploded = dfp_train_select.explode('cards')\ndfp_train_select_exploded.rename(mapper={\"cards\" : \"card\"}, axis=1, inplace=True)\nprint(\"Count of rows (dfp_train_select_exploded):\", len(dfp_train_select_exploded))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:38.200148Z","iopub.execute_input":"2022-07-26T23:10:38.200437Z","iopub.status.idle":"2022-07-26T23:10:38.230561Z","shell.execute_reply.started":"2022-07-26T23:10:38.200413Z","shell.execute_reply":"2022-07-26T23:10:38.229494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfp_agg = dfp_train_select_exploded.groupby([ 'hero', 'is_wild', 'is_standard', 'card']).size().to_frame().reset_index()\ndfp_agg.columns = ['hero', 'is_wild', 'is_standard', 'card', 'count']\n\ndfp_agg[\"hero-is_wild-is_standard\"] = dfp_agg.apply(lambda row: f\"{row['hero']}-{row['is_wild']}-{row['is_standard']}\", axis=1)\ndfp_agg[\"dict_card_count\"] = dfp_agg.apply(lambda row: {\"card\" : row['card'], \"count\" : row['count']}, axis=1)\n\ndfp_agg = dfp_agg.groupby([\"hero-is_wild-is_standard\"])['dict_card_count'].apply(list).to_frame()\ndfp_agg.reset_index(inplace=True)\n\ndef rank_card(dict_card_count):\n    return pd.DataFrame(dict_card_count).sort_values(\"count\", ascending=False)[\"card\"].tolist()\n\ndfp_agg[\"ranking\"] = dfp_agg['dict_card_count'].apply(lambda dict_card_count: rank_card(dict_card_count))\ndict_ranking = dfp_agg[[\"hero-is_wild-is_standard\", \"ranking\"]].set_index(\"hero-is_wild-is_standard\").to_dict(orient=\"index\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:38.231853Z","iopub.execute_input":"2022-07-26T23:10:38.232768Z","iopub.status.idle":"2022-07-26T23:10:38.361805Z","shell.execute_reply.started":"2022-07-26T23:10:38.232735Z","shell.execute_reply":"2022-07-26T23:10:38.360564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncardids = dfp_cards[\"id\"].tolist()\ndef build_recommendations(hero, is_wild, is_standard, cards, dict_ranking, k=3):\n    deck_incomplete = literal_eval(cards)\n    key_ranking = f\"{hero}-{is_wild}-{is_standard}\"\n    recommendations = []\n    if key_ranking in dict_ranking:\n        ranked_cards = dict_ranking[key_ranking][\"ranking\"]\n        for elt in ranked_cards:\n            if deck_incomplete.count(elt) < 2:\n                recommendations.append(elt)\n                \n            if len(recommendations) == k:\n                break\n    \n    if len(recommendations) < k:\n        recommendations.extend(random.choices(cardids, k=k-len(recommendations)))\n        \n    return \" \".join([str(elt) for elt in recommendations])\n\ndfp_test[\"recommendations\"] = dfp_test.apply(lambda row: build_recommendations(row[\"hero\"], row[\"is_wild\"], row[\"is_standard\"], row[\"cards_incomplete\"], dict_ranking, k=3), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:38.363736Z","iopub.execute_input":"2022-07-26T23:10:38.364091Z","iopub.status.idle":"2022-07-26T23:10:38.649633Z","shell.execute_reply.started":"2022-07-26T23:10:38.364062Z","shell.execute_reply":"2022-07-26T23:10:38.648612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfp_submission = dfp_test[[\"deckid\", \"recommendations\"]].copy()\ndfp_submission.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T23:10:38.650846Z","iopub.execute_input":"2022-07-26T23:10:38.651185Z","iopub.status.idle":"2022-07-26T23:10:38.672366Z","shell.execute_reply.started":"2022-07-26T23:10:38.651151Z","shell.execute_reply":"2022-07-26T23:10:38.671239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}