{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"filename = \"../input/train.csv\"\n\nn = sum(1 for line in open(filename)) - 1 #number of records in file (excludes header)\nprint(\"number of trainset record in total is :\", n)\n\ns = 100000 #desired sample size\nskip = sorted(random.sample(range(1,n+1),n-s)) #the 0-indexed header will not be included in the skip list\n\ntrain = pd.read_csv(filename, parse_dates=['srch_ci', 'srch_co'], skiprows=skip)\ntrain.info()"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"filename = \"../input/test.csv\"\n\nm = sum(1 for line in open(filename)) - 1 #number of records in file (excludes header)\nprint(\"number of testset records in total is :\", m)\n\ns = int(100000 * m /n) #desired sample size\nskip = sorted(random.sample(range(1,m+1),m-s)) #the 0-indexed header will not be included in the skip list\n\ntest = pd.read_csv(filename, parse_dates=['srch_ci', 'srch_co'], skiprows=skip)\ntest.info()"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# preferred continent destinations\nsns.countplot(x='hotel_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"sns.countplot(x='hotel_continent', data=test)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"# most of people booking are from continent 3 I guess is one of the rich continent?\nsns.countplot(x='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"sns.countplot(x='posa_continent', data=test)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"sns.countplot(x='hotel_continent', hue='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"sns.countplot(x='hotel_continent', hue='posa_continent', data=test)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":0}