{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom datetime import datetime","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"chunk_train_df = pd.read_csv(\"../input/expedia-hotel-recommendations/train.csv\", chunksize=100000)\ntest_df = pd.read_csv(\"../input/expedia-hotel-recommendations/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def chunk_preprocessing(temp_chunk):\n    dropped_columns = [\"site_name\", \n                       \"posa_continent\", \n                       \"is_mobile\", \n                       \"is_package\", \n                       \"channel\", \n                       \"srch_destination_type_id\",\n                       \"cnt\",\n                       \"hotel_continent\",\n                       \"hotel_market\"\n                      ]\n    temp_chunk = temp_chunk.loc[temp_chunk['is_booking'] == 1]\n    temp_chunk = temp_chunk.drop(dropped_columns, axis=1)\n    return temp_chunk.dropna().drop_duplicates()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"chunk_list = []  # append each chunk df here \n\n# Each chunk is in df format\nfor chunk in chunk_train_df:  \n    # perform data filtering \n    chunk_filter = chunk_preprocessing(chunk)\n    \n    # Once the data filtering is done, append the chunk to list\n    chunk_list.append(chunk_filter)\n    \n# concat the list into dataframe \ntrain_df = pd.concat(chunk_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_year(x):\n    if x is not None and type(x) is not float:\n        try:\n            return datetime.strptime(x, '%Y-%m-%d').year\n        except ValueError:\n            try:\n                return datetime.strptime(x, '%Y-%m-%d %H:%M:%S').year\n            except ValueError:\n                pass\n    else:\n        return 2013\n    pass\n\ndef get_month(x):\n    if x is not None and type(x) is not float:\n        try:\n            return datetime.strptime(x, '%Y-%m-%d').month\n        except:\n            try:\n                return datetime.strptime(x, '%Y-%m-%d %H:%M:%S').month\n            except ValueError:\n                pass\n    else:\n        return 1\n    pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['date_time_year'] = pd.Series(train_df.date_time, index = train_df.index)\ntrain_df['date_time_month'] = pd.Series(train_df.date_time, index = train_df.index)\ntrain_df.date_time_year = train_df.date_time_year.apply(lambda x: get_year(x))\ntrain_df.date_time_month = train_df.date_time_month.apply(lambda x: get_month(x))\ndel train_df['date_time']\n\ntrain_df['srch_ci_year'] = pd.Series(train_df.srch_ci, index=train_df.index)\ntrain_df['srch_ci_month'] = pd.Series(train_df.srch_ci, index=train_df.index)\ntrain_df.srch_ci_year = train_df.srch_ci_year.apply(lambda x: get_year(x))\ntrain_df.srch_ci_month = train_df.srch_ci_month.apply(lambda x: get_month(x))\ndel train_df['srch_ci']\n\ntrain_df['srch_co_year'] = pd.Series(train_df.srch_co, index=train_df.index)\ntrain_df['srch_co_month'] = pd.Series(train_df.srch_co, index=train_df.index)\ntrain_df.srch_co_year = train_df.srch_co_year.apply(lambda x: get_year(x))\ntrain_df.srch_co_month = train_df.srch_co_month.apply(lambda x: get_month(x))\ndel train_df['srch_co']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df['date_time_year'] = pd.Series(test_df.date_time, index = test_df.index)\ntest_df['date_time_month'] = pd.Series(test_df.date_time, index = test_df.index)\ntest_df.date_time_year = test_df.date_time_year.apply(lambda x: get_year(x))\ntest_df.date_time_month = test_df.date_time_month.apply(lambda x: get_month(x))\ndel test_df['date_time']\n\ntest_df['srch_ci_year'] = pd.Series(test_df.srch_ci, index=test_df.index)\ntest_df['srch_ci_month'] = pd.Series(test_df.srch_ci, index=test_df.index)\ntest_df.srch_ci_year = test_df.srch_ci_year.apply(lambda x: get_year(x))\ntest_df.srch_ci_month = test_df.srch_ci_month.apply(lambda x: get_month(x))\ndel test_df['srch_ci']\n\ntest_df['srch_co_year'] = pd.Series(test_df.srch_co, index=test_df.index)\ntest_df['srch_co_month'] = pd.Series(test_df.srch_co, index=test_df.index)\ntest_df.srch_co_year = test_df.srch_co_year.apply(lambda x: get_year(x))\ntest_df.srch_co_month = test_df.srch_co_month.apply(lambda x: get_month(x))\ndel test_df['srch_co']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = test_df.drop([\"site_name\", \n                       \"posa_continent\", \n                       \"is_mobile\", \n                       \"is_package\", \n                       \"channel\", \n                       \"srch_destination_type_id\",\n                       \"hotel_continent\",\n                       \"hotel_market\"\n                      ], axis=1).dropna().drop_duplicates()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_df.shape)\nprint(test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.to_csv(\"train_data_mod.csv\", index=False)\ntest_df.to_csv(\"test_data_mod.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"End\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}