{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \nimport pandas as pd\nimport datetime\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\nimport os\nimport sys\nimport pickle\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/working'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-30T02:36:19.649323Z","iopub.execute_input":"2023-04-30T02:36:19.649714Z","iopub.status.idle":"2023-04-30T02:36:22.698331Z","shell.execute_reply.started":"2023-04-30T02:36:19.649646Z","shell.execute_reply":"2023-04-30T02:36:22.697251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dateparse = lambda x: pd.datetime.strptime(str(x), '%Y-%m-%d') if str(x)!=\"nan\" else \"0\"\ntrain = pd.read_csv(\"/kaggle/input/expedia-hotel-recommendations/train.csv\", nrows=95000, date_parser=dateparse).dropna()","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-04-30T03:09:53.460203Z","iopub.execute_input":"2023-04-30T03:09:53.460881Z","iopub.status.idle":"2023-04-30T03:09:53.700288Z","shell.execute_reply.started":"2023-04-30T03:09:53.460827Z","shell.execute_reply":"2023-04-30T03:09:53.698949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"user_id\"].shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T03:09:54.302406Z","iopub.execute_input":"2023-04-30T03:09:54.302749Z","iopub.status.idle":"2023-04-30T03:09:54.309627Z","shell.execute_reply.started":"2023-04-30T03:09:54.302688Z","shell.execute_reply":"2023-04-30T03:09:54.308204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_tests = train[\"user_id\"].unique()\nunique_tests.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T03:09:54.898148Z","iopub.execute_input":"2023-04-30T03:09:54.898476Z","iopub.status.idle":"2023-04-30T03:09:54.905738Z","shell.execute_reply.started":"2023-04-30T03:09:54.898427Z","shell.execute_reply":"2023-04-30T03:09:54.904758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs, outputs = [], []\nfrom tqdm import tqdm\nfor ind in tqdm(train.index):\n    ci = train[\"srch_ci\"][ind]\n    co = train[\"srch_co\"][ind]\n    ci = datetime.strptime(ci, '%Y-%m-%d')\n    co = datetime.strptime(co, '%Y-%m-%d')\n    odd = train['orig_destination_distance'][ind]    \n    date_diff = abs(pd.Timedelta(co - ci).days)\n    user_location_countries = train[\"user_location_country\"][ind]\n    user_location_regions = train[\"user_location_region\"][ind]\n    user_location_cities = train[\"user_location_city\"][ind]\n    srch_destination_type_ids = train[\"srch_destination_type_id\"][ind]\n    hotel_continents = train[\"hotel_continent\"][ind]\n    hotel_countries = train[\"hotel_country\"][ind]\n    hotel_cluster = train['hotel_cluster'][ind]\n    hotel_markets = train[\"hotel_market\"][ind]\n    srch_adults_cnts = train[\"srch_adults_cnt\"][ind]\n    srch_children_cnts = train[\"srch_children_cnt\"][ind]\n    srch_rm_cnts = train[\"srch_rm_cnt\"][ind]\n    stay = date_diff\n    user_ids = train[\"user_id\"][ind]\n    srch_destination_ids = train[\"srch_destination_id\"][ind]\n    d1 =  ci\n    did = train['srch_destination_id'][ind]\n    isweekend = 0\n    if d1.weekday() > 3:\n        isweekend = 1\n    ismonthend = 0\n    if abs(d1.day - 30)< 3 or abs(d1.day - 1)<3:\n        ismonthend = 1\n    is_summer=0\n    is_winter=0\n    is_spring=0\n    is_rainy=0\n    mo = d1.month\n    da = d1.day\n    if mo>1 and mo<4:\n        is_spring = 1\n    elif mo<7:\n        is_summer = 1\n    elif mo<10:\n        is_rainy = 1\n    else:\n        is_winter = 1\n    adults = train['srch_adults_cnt'][ind]\n    children = train['srch_children_cnt'][ind]\n    rm = train['srch_rm_cnt'][ind]\n    is_family = 0\n    if adults>0 and children>0:\n        is_family = 1\n    is_trip = 0\n    if rm>5:\n        is_trip = 1\n    is_valentines = 0\n    is_christmas = 0\n    if mo==2 and abs(da-14)<2:\n        is_valentines = 1\n    if mo==12 and abs(da-25)<5:\n        is_christmas = 1\n    vector = [odd, user_location_countries,user_location_regions,user_location_cities,srch_destination_type_ids,hotel_continents, hotel_countries,hotel_markets,srch_adults_cnts,srch_children_cnts,srch_rm_cnts,stay,user_ids,srch_destination_ids,isweekend,ismonthend, is_valentines,is_christmas,\n       is_spring,is_summer,is_rainy,is_winter,is_family,is_trip]\n    inputs.append(vector)\n    outputs.append(hotel_cluster)\ninputs = np.array(inputs) \noutputs = np.array(outputs)\ninputs.shape, outputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T03:09:55.711121Z","iopub.execute_input":"2023-04-30T03:09:55.711465Z","iopub.status.idle":"2023-04-30T03:10:18.858262Z","shell.execute_reply.started":"2023-04-30T03:09:55.711415Z","shell.execute_reply":"2023-04-30T03:10:18.857174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import AdaBoostClassifier\n\ndef mapKeval(preds, actual, k=10):\n    predicted = preds.argsort(axis=1)[:,-np.arange(k)]\n    metric = 0.\n    for i in range(k):\n        metric += np.sum(actual==predicted[:,i])/(i+1)\n    metric /= actual.shape[0]\n    return 'MAP@5', metric\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=23)\nmapks = []\nfor n_fold, (train_idx, valid_idx) in enumerate(skf.split(inputs,outputs)):\n    X_train, y_train = inputs[train_idx], outputs[train_idx]\n    X_valid, y_valid = inputs[valid_idx], outputs[valid_idx]\n#     clf = AdaBoostClassifier(n_estimators=100, base_estimator=DecisionTreeClassifier(max_depth=31))\n#     clf = DecisionTreeClassifier(max_depth=31)\n#     clf = KNeighborsClassifier(n_neighbors=3)\n    clf = RandomForestClassifier(max_depth=31, n_estimators=300,n_jobs=-1,warm_start=True)\n    clf.fit(X_train, y_train)\n    preds = clf.predict_proba(X_valid)\n    mapks.append(mapKeval(preds, y_valid, k=10)[1])\n    print(\"Fold:\",n_fold, \"Map@10:\", mapks[-1])","metadata":{"execution":{"iopub.status.busy":"2023-04-30T03:10:34.421831Z","iopub.execute_input":"2023-04-30T03:10:34.422358Z","iopub.status.idle":"2023-04-30T03:12:51.909793Z","shell.execute_reply.started":"2023-04-30T03:10:34.422312Z","shell.execute_reply":"2023-04-30T03:12:51.908486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(mapks) / len(mapks)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T03:12:51.912052Z","iopub.execute_input":"2023-04-30T03:12:51.912456Z","iopub.status.idle":"2023-04-30T03:12:51.920325Z","shell.execute_reply.started":"2023-04-30T03:12:51.912382Z","shell.execute_reply":"2023-04-30T03:12:51.918876Z"},"trusted":true},"execution_count":null,"outputs":[]}]}