{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-29T10:49:52.566608Z","iopub.execute_input":"2023-04-29T10:49:52.567553Z","iopub.status.idle":"2023-04-29T10:49:52.583046Z","shell.execute_reply.started":"2023-04-29T10:49:52.567505Z","shell.execute_reply":"2023-04-29T10:49:52.581806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the datas \nimport pandas as pd\ndestinations = pd.read_csv(\"/kaggle/input/expedia-hotel-recommendations/destinations.csv\")\ntest = pd.read_csv(\"/kaggle/input/expedia-hotel-recommendations/test.csv\")\ntrain = pd.read_csv(\"/kaggle/input/expedia-hotel-recommendations/train.csv\", nrows=5000000)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:49:52.585203Z","iopub.execute_input":"2023-04-29T10:49:52.585827Z","iopub.status.idle":"2023-04-29T10:50:14.465357Z","shell.execute_reply.started":"2023-04-29T10:49:52.585791Z","shell.execute_reply":"2023-04-29T10:50:14.464436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understanding the datas\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.466670Z","iopub.execute_input":"2023-04-29T10:50:14.467250Z","iopub.status.idle":"2023-04-29T10:50:14.474964Z","shell.execute_reply.started":"2023-04-29T10:50:14.467212Z","shell.execute_reply":"2023-04-29T10:50:14.473595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.476601Z","iopub.execute_input":"2023-04-29T10:50:14.477422Z","iopub.status.idle":"2023-04-29T10:50:14.488748Z","shell.execute_reply.started":"2023-04-29T10:50:14.477374Z","shell.execute_reply":"2023-04-29T10:50:14.487104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.492434Z","iopub.execute_input":"2023-04-29T10:50:14.492889Z","iopub.status.idle":"2023-04-29T10:50:14.519677Z","shell.execute_reply.started":"2023-04-29T10:50:14.492824Z","shell.execute_reply":"2023-04-29T10:50:14.518340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.521315Z","iopub.execute_input":"2023-04-29T10:50:14.521688Z","iopub.status.idle":"2023-04-29T10:50:14.545357Z","shell.execute_reply.started":"2023-04-29T10:50:14.521652Z","shell.execute_reply":"2023-04-29T10:50:14.544167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"・date_timeが結構重要で、テストデータが2015年、訓練データが2013-14年の話になっている\n　→user_idがかぶっている\n","metadata":{}},{"cell_type":"code","source":"# Figuring out what to predict\n# First, i predict hotel_cluster for scoring this competition\n\n# Exploring hotel clusters\ntrain[\"hotel_cluster\"].value_counts()\n\n# There doesn't appear to be any relationship between cluster number and the number of items","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.547088Z","iopub.execute_input":"2023-04-29T10:50:14.547858Z","iopub.status.idle":"2023-04-29T10:50:14.600636Z","shell.execute_reply.started":"2023-04-29T10:50:14.547784Z","shell.execute_reply":"2023-04-29T10:50:14.599250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exploring train and test user ids\n# Setting unique values\ntest_ids = set(test.user_id.unique())\ntrain_ids = set(train.user_id.unique())\nintersection_count = len(test_ids & train_ids)\nintersection_count == len(test_ids)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.602435Z","iopub.execute_input":"2023-04-29T10:50:14.602927Z","iopub.status.idle":"2023-04-29T10:50:14.962131Z","shell.execute_reply.started":"2023-04-29T10:50:14.602886Z","shell.execute_reply":"2023-04-29T10:50:14.960786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"一応ここでTrueになるらしい。\nUser_idをユニークにしたとき、テストデータと訓練データで数が同じになる。","metadata":{}},{"cell_type":"code","source":"# Add in times and dates\n# Changing to a datetime value & separate \"year\", \"month\"\ntrain[\"date_time\"] = pd.to_datetime(train[\"date_time\"])\ntrain[\"year\"] = train[\"date_time\"].dt.year\ntrain[\"month\"] = train[\"date_time\"].dt.month","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:14.963592Z","iopub.execute_input":"2023-04-29T10:50:14.963995Z","iopub.status.idle":"2023-04-29T10:50:17.046551Z","shell.execute_reply.started":"2023-04-29T10:50:14.963956Z","shell.execute_reply":"2023-04-29T10:50:17.045452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick 10000 users\nimport random\n\nunique_users = train.user_id.unique()\nunique_user_id = unique_users.tolist()\n\nsel_user_id = random.sample(unique_user_id, 100000)\nsel_train = train[train.user_id.isin(sel_user_id)]","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:17.047957Z","iopub.execute_input":"2023-04-29T10:50:17.048410Z","iopub.status.idle":"2023-04-29T10:50:18.398925Z","shell.execute_reply.started":"2023-04-29T10:50:17.048373Z","shell.execute_reply":"2023-04-29T10:50:18.397468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick new training and testing sets\n# Reproducing original train, test \nt1 = sel_train[((sel_train.year == 2013) | ((sel_train.year == 2014) & (sel_train.month < 8)))]\nt2 = sel_train[((sel_train.year == 2014) & (sel_train.month >= 8))]","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:18.402304Z","iopub.execute_input":"2023-04-29T10:50:18.403218Z","iopub.status.idle":"2023-04-29T10:50:18.758178Z","shell.execute_reply.started":"2023-04-29T10:50:18.403161Z","shell.execute_reply":"2023-04-29T10:50:18.756866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove click events\n# Leaving only 1\nt2 = t2[t2.is_booking == True]\n","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:18.759753Z","iopub.execute_input":"2023-04-29T10:50:18.760164Z","iopub.status.idle":"2023-04-29T10:50:18.803372Z","shell.execute_reply.started":"2023-04-29T10:50:18.760128Z","shell.execute_reply":"2023-04-29T10:50:18.802210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A simple algorithm\nmost_common_clusters = list(train.hotel_cluster.value_counts().head().index)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:18.804801Z","iopub.execute_input":"2023-04-29T10:50:18.805228Z","iopub.status.idle":"2023-04-29T10:50:18.851896Z","shell.execute_reply.started":"2023-04-29T10:50:18.805195Z","shell.execute_reply":"2023-04-29T10:50:18.850911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating predictions\npredictions = [most_common_clusters for i in range(t2.shape[0])]\n","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:18.855917Z","iopub.execute_input":"2023-04-29T10:50:18.857207Z","iopub.status.idle":"2023-04-29T10:50:18.870369Z","shell.execute_reply.started":"2023-04-29T10:50:18.857161Z","shell.execute_reply":"2023-04-29T10:50:18.869148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating error\n# Trying map@5\ntarget = [[l] for l in t2[\"hotel_cluster\"]]\nfrom tqdm.auto import tqdm\n\n# Oh, i cant use ml_metrics...\n# Reference of map@k: https://github.com/benhamner/Metrics/blob/master/Python/ml_metrics/average_precision.py\n\ndef apk(actual, predicted, k=10):\n    \"\"\"\n    Computes the average precision at k.\n    This function computes the average prescision at k between two lists of\n    items.\n    Parameters\n    ----------\n    actual : list\n             A list of elements that are to be predicted (order doesn't matter)\n    predicted : list\n                A list of predicted elements (order does matter)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The average precision at k over the input lists\n    \"\"\"\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    # remove this case in advance\n    # if not actual:\n    #     return 0.0\n\n    return score / min(len(actual), k)\n\n\ndef mapk(actual, predicted, k=10):\n    \"\"\"\n    Computes the mean average precision at k.\n    This function computes the mean average prescision at k between two lists\n    of lists of items.\n    Parameters\n    ----------\n    actual : list\n             A list of lists of elements that are to be predicted \n             (order doesn't matter in the lists)\n    predicted : list\n                A list of lists of predicted elements\n                (order matters in the lists)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The mean average precision at k over the input lists\n    \"\"\"\n    return np.mean([apk(a,p,k) for a,p in zip(actual, predicted)])\n\ntqdm.pandas()\n\nmapk(\n    target, \n    predictions, \n    k=5\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:18.872316Z","iopub.execute_input":"2023-04-29T10:50:18.872694Z","iopub.status.idle":"2023-04-29T10:50:19.076999Z","shell.execute_reply.started":"2023-04-29T10:50:18.872659Z","shell.execute_reply":"2023-04-29T10:50:19.075317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find correlations\ntrain.corr()[\"hotel_cluster\"]\n\n# NO columns correlate linerly with hotel_cluster...","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:19.079411Z","iopub.execute_input":"2023-04-29T10:50:19.079987Z","iopub.status.idle":"2023-04-29T10:50:27.155681Z","shell.execute_reply.started":"2023-04-29T10:50:19.079934Z","shell.execute_reply":"2023-04-29T10:50:27.154276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating features from destinations\n# Using PCA\n\nfrom sklearn.decomposition import PCA\n\npca = PCA(n_components=3)\ndest_small = pca.fit_transform(destinations[[\"d{0}\".format(i + 1) for i in range(149)]])\ndest_small = pd.DataFrame(dest_small)\ndest_small[\"srch_destination_id\"] = destinations[\"srch_destination_id\"]","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:27.157247Z","iopub.execute_input":"2023-04-29T10:50:27.158066Z","iopub.status.idle":"2023-04-29T10:50:27.847815Z","shell.execute_reply.started":"2023-04-29T10:50:27.158023Z","shell.execute_reply":"2023-04-29T10:50:27.845941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating features\n# Generate new date features based on date_time, srch_ci, and srch_co.\n# Remove non-numeric columns like date_time.\n# Add in features from dest_small.\n# Replace any missing values with -1.\n\ndef calc_fast_features(df):\n    df[\"date_time\"] = pd.to_datetime(df[\"date_time\"])\n    df[\"srch_ci\"] = pd.to_datetime(df[\"srch_ci\"], format='%Y-%m-%d', errors=\"coerce\")\n    df[\"srch_co\"] = pd.to_datetime(df[\"srch_co\"], format='%Y-%m-%d', errors=\"coerce\")\n\n    props = {}\n    for prop in [\"month\", \"day\", \"hour\", \"minute\", \"dayofweek\", \"quarter\"]:\n        props[prop] = getattr(df[\"date_time\"].dt, prop)\n\n    carryover = [p for p in df.columns if p not in [\"date_time\", \"srch_ci\", \"srch_co\"]]\n    for prop in carryover:\n        props[prop] = df[prop]\n\n    date_props = [\"month\", \"day\", \"dayofweek\", \"quarter\"]\n    for prop in date_props:\n        props[\"ci_{0}\".format(prop)] = getattr(df[\"srch_ci\"].dt, prop)\n        props[\"co_{0}\".format(prop)] = getattr(df[\"srch_co\"].dt, prop)\n    props[\"stay_span\"] = (df[\"srch_co\"] - df[\"srch_ci\"]).astype('timedelta64[h]')\n\n    ret = pd.DataFrame(props)\n\n    ret = ret.join(dest_small, on=\"srch_destination_id\", how='left', rsuffix=\"dest\")\n    ret = ret.drop(\"srch_destination_iddest\", axis=1)\n    return ret\n\ndf = calc_fast_features(t1)\ndf.fillna(-1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:27.851862Z","iopub.execute_input":"2023-04-29T10:50:27.853165Z","iopub.status.idle":"2023-04-29T10:50:33.012326Z","shell.execute_reply.started":"2023-04-29T10:50:27.853086Z","shell.execute_reply":"2023-04-29T10:50:33.011092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Machine learning\n\npredictors = [c for c in df.columns if c not in [\"hotel_cluster\"]]\nfrom sklearn.ensemble import RandomForestClassifier\nclf = RandomForestClassifier(n_estimators=10, min_weight_fraction_leaf=0.1)\n\n# Not so good accuracy","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:33.014054Z","iopub.execute_input":"2023-04-29T10:50:33.014416Z","iopub.status.idle":"2023-04-29T10:50:33.020214Z","shell.execute_reply.started":"2023-04-29T10:50:33.014380Z","shell.execute_reply":"2023-04-29T10:50:33.019272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Top clusters based on hotel_cluster\n# Find the most popular hotel clusters for each destination\n# Weight bookings higher than clicks\ndef make_key(items):\n    return \"_\".join([str(i) for i in items])\n\nmatch_cols = [\"srch_destination_id\"]\ncluster_cols = match_cols + ['hotel_cluster']\ngroups = t1.groupby(cluster_cols)\ntop_clusters = {}\nfor name, group in groups:\n    clicks = len(group.is_booking[group.is_booking == False])\n    bookings = len(group.is_booking[group.is_booking == True])\n\n    score = bookings + .15 * clicks\n\n    clus_name = make_key(name[:len(match_cols)])\n    if clus_name not in top_clusters:\n        top_clusters[clus_name] = {}\n    top_clusters[clus_name][name[-1]] = score","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:50:33.021613Z","iopub.execute_input":"2023-04-29T10:50:33.022661Z","iopub.status.idle":"2023-04-29T10:51:50.422969Z","shell.execute_reply.started":"2023-04-29T10:50:33.022623Z","shell.execute_reply":"2023-04-29T10:51:50.421622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform the dictionary to find the top 5 hotel clusters for each srch_destination_id\nimport operator\n\ncluster_dict = {}\nfor n in top_clusters:\n    tc = top_clusters[n]\n    top = [l[0] for l in sorted(tc.items(), key=operator.itemgetter(1), reverse=True)[:5]]\n    cluster_dict[n] = top","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:51:50.424412Z","iopub.execute_input":"2023-04-29T10:51:50.424767Z","iopub.status.idle":"2023-04-29T10:51:50.509560Z","shell.execute_reply.started":"2023-04-29T10:51:50.424731Z","shell.execute_reply":"2023-04-29T10:51:50.508282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making predictions based on destination\n# Iterate through each row in t2.\n# Extract the srch_destination_id for the row.\n# Find the top clusters for that destination id.\n# Append the top clusters to preds.\n\npreds = []\nfor index, row in t2.iterrows():\n    key = make_key([row[m] for m in match_cols])\n    if key in cluster_dict:\n        preds.append(cluster_dict[key])\n    else:\n        preds.append([])","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:51:50.511515Z","iopub.execute_input":"2023-04-29T10:51:50.512045Z","iopub.status.idle":"2023-04-29T10:51:55.822732Z","shell.execute_reply.started":"2023-04-29T10:51:50.511995Z","shell.execute_reply":"2023-04-29T10:51:55.821174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute accuracy using the mapk function \nmapk([[l] for l in t2[\"hotel_cluster\"]], preds, k=5)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:51:55.824336Z","iopub.execute_input":"2023-04-29T10:51:55.824809Z","iopub.status.idle":"2023-04-29T10:51:56.068366Z","shell.execute_reply.started":"2023-04-29T10:51:55.824760Z","shell.execute_reply":"2023-04-29T10:51:56.067062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"match_cols = ['user_location_country', 'user_location_region', 'user_location_city', 'hotel_market', 'orig_destination_distance']\n\ngroups = t1.groupby(match_cols)\n\ndef generate_exact_matches(row, match_cols):\n    index = tuple([row[t] for t in match_cols])\n    try:\n        group = groups.get_group(index)\n    except Exception:\n        return []\n    clus = list(set(group.hotel_cluster))\n    return clus\n\nexact_matches = []\nfor i in range(t2.shape[0]):\n    exact_matches.append(generate_exact_matches(t2.iloc[i], match_cols))","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:51:56.069729Z","iopub.execute_input":"2023-04-29T10:51:56.070115Z","iopub.status.idle":"2023-04-29T10:52:25.604844Z","shell.execute_reply.started":"2023-04-29T10:51:56.070044Z","shell.execute_reply":"2023-04-29T10:52:25.602953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def f5(seq, idfun=None):\n    if idfun is None:\n        def idfun(x): return x\n    seen = {}\n    result = []\n    for item in seq:\n        marker = idfun(item)\n        if marker in seen: continue\n        seen[marker] = 1\n        result.append(item)\n    return result\n\nfull_preds = [f5(exact_matches[p] + preds[p] + most_common_clusters)[:5] for p in range(len(preds))]\nmapk([[l] for l in t2[\"hotel_cluster\"]], full_preds, k=5)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:52:25.607039Z","iopub.execute_input":"2023-04-29T10:52:25.608019Z","iopub.status.idle":"2023-04-29T10:52:26.608654Z","shell.execute_reply.started":"2023-04-29T10:52:25.607961Z","shell.execute_reply":"2023-04-29T10:52:26.607106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission\n\nt1 = train\nt2 = test\nwrite_p = [\" \".join([str(l) for l in p]) for p in full_preds]\nwrite_frame = [\"{0},{1}\".format(t2[\"id\"][i], write_p[i]) for i in range(len(full_preds))]\nwrite_frame = [\"id\",\"hotel_clusters\"] + write_frame\nwith open(\"predictions.csv\", \"w+\") as f:\n    f.write(\"\\n\".join(write_frame))","metadata":{"execution":{"iopub.status.busy":"2023-04-29T10:52:26.610540Z","iopub.execute_input":"2023-04-29T10:52:26.611062Z","iopub.status.idle":"2023-04-29T10:52:27.516246Z","shell.execute_reply.started":"2023-04-29T10:52:26.611010Z","shell.execute_reply":"2023-04-29T10:52:27.515170Z"},"trusted":true},"execution_count":null,"outputs":[]}]}