{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('max_columns', None)\ndata_chunk = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', parse_dates=['date_time', 'srch_ci', 'srch_co'], nrows=10**6)\n\n\ndata_chunk.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_chunk.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_chunk.dtypes.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compress_dataset(data_chunk):\n    for feature, data_type in zip(data_chunk.dtypes.index, data_chunk.dtypes.values):\n    #     print(feature, data_type)\n        if data_type == 'int64':\n            data_chunk[feature] = data_chunk[feature].astype('int8')\n        if data_type=='float64':\n            data_chunk[feature] = data_chunk[feature].astype('float16')\n    return data_chunk\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compressed_data = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', iterator=True) as reader:\n    data = reader.get_chunk(10**6)\n    \n    while data is not None:\n        \n    #     print(data.info(verbose=False))\n        data = compress_dataset(data)\n    #     print(data.info(verbose=False))\n        compressed_data=compressed_data.append(data)\n        \n        try: \n            data = reader.get_chunk(10**6)\n        except:\n            print('Done reading hwole dataset!!')\n            break\n            ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compressed_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='posa_continent', data=compressed_data)             ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compressed_data['site_name'].value_counts(ascending=False)[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compressed_data['hotel_cluster'].value_counts(ascending=False)[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"destinations = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/destinations.csv')\n\ndestinations.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', iterator=True) as reader:\n    data = reader.get_chunk(10**6)\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndata.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"destinations.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" The original data consists of 37 million rows which represent 1.2 million users, so its better to downsample it as far as possible","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t1 = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', iterator=True) as reader:\n    data = reader.get_chunk(10**4)\n    t1 = data[data.is_booking != 0]\n    \nt1.groupby('user_id').size().unstack()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t1 = data[data.is_booking != 0]\ngroupby","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pda\nprocessed_data = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', iterator=True) as reader:\n    data = reader.get_chunk(10**6)\n    chunks=1\n    while data is not None:\n        t1 = data[data.is_booking != 0]\n        unique_users = t1.user_id.unique()\n        \n        t1.groupby('user_id').count\n        \n        for user in unique_users:\n            bookings = len(t1.loc[t1['user_id'] == user])\n            if bookings>=20:\n                t1 = t1[t1.user_id != user]\n        print(t1.shape)\n        print('Proccessed chunk number : {}'.format(chunks))\n        processed_data = processed_data.append(t1)\n        break\n        try:\n            data = reader.get_chunk(10**6)\n        except :\n            print('Read all data!!')\n            break\n            \n        chunks+=1\n    print(chunks)\n    \n    \n                \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,8))a\nsns.set(style=\"ticks\", font_scale = 1)\nax = sns.countplot(data = data,x='site_name',order = data['Class'].value_counts().index,palette=\"flare\")\ns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyNumbers:\n  def __iter__(self):\n    self.a = 1\n    return self\n\n  def __next__(self):\n    if self.a <= 20:\n      x = self.a\n      self.a += 1\n      return x\n    else:\n      raise StopIteration\n\nmyclass = MyNumbers()\nmyiter = iter(myclass)\n\nfor x in myiter:\n  print(x)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}