{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# HANDLE DATA SIZE\n# Downcaster les colonnes int64 \n\ndef handle_data(data):\n\n    list_int64 = data.select_dtypes(include=['int64']).columns.tolist()\n\n    for col in list_int64 :\n        if data[col].max() <= 255 and data[col].min() > 0  :\n            data[col] = data[col].astype('uint8')  # les uint8 rassemble les entiers positifs 0 à 255 \n        elif data[col].max() <= 65535 and data[col].min() > 0  :\n            data[col] = data[col].astype('uint16')\n        elif data[col].max() <= 4294967295 and data[col].min() > 0  :\n            data[col] = data[col].astype('uint32')\n        elif data[col].max() <= 127 and data[col].min() >= -128  :\n            data[col] = data[col].astype('int8')\n        elif data[col].max() <= 32767 and data[col].min() >= -32768  :\n            data[col] = data[col].astype('int16')\n\n    # transfomer les colonnes str en variables catégorielles\n    list_str = data.select_dtypes(include=['object']).columns.tolist()\n    for col in list_str :\n        data[col] = data[col].astype('category') \n\n    dict_dtypes = data.dtypes.to_dict()\n    return dict_dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_rows = 3767000\ntest_rows = 2530000\n\ndata_train_init = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', nrows = train_rows*0.001)\ndata_test_init = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/test.csv', nrows = test_rows*0.001)\n\ndtypes_train = handle_data(data_train_init)\ndtypes_test =handle_data(data_test_init)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dtypes_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import random\n    \ndef get_random_idx(liste_idx) :\n    random.seed(42)\n    sample = random.sample(liste_idx, k=int(len(liste_idx)*0.7))\n    print(int(len(liste_idx)*0.7))\n    return sample","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_rows = 3767000\ntest_rows = 2530000\n\nliste_idx_train = range(3767000)\nliste_idx_test = range(2530000)\n\nlist_idx_random_train = get_random_idx(liste_idx_train)\nlist_idx_random_test = get_random_idx(liste_idx_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', dtype=dtypes_train, nrows = 300_000)\n#data_test = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/test.csv', dtype=dtypes_test, skiprows = list_idx_random_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = data_train.iloc[:, 18].values  #train on is_booking -> booké ou non\ny = data_train.iloc[:, -1].values  #test on hotel_cluster -> quel type d'hotel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = X.reshape(-1,1)\ny","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.unique(y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.25, random_state = 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nclassifier = DecisionTreeClassifier(criterion='entropy', random_state=0)\nclassifier.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = classifier.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classifier.score(X_test, y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import accuracy_score\naccuracy_score(y_test, y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import f1_score\n\nf1_weighted = f1_score(y_test, y_pred, average='weighted')\nf1_macro = f1_score(y_test, y_pred, average='macro')\n\nprint(\"weighted :\", f1_weighted, \"\\n--------------------------------\\n\",\"macro :\", f1_macro)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_test, y_pred)\ncm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.tree import export_graphviz","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pip install pydotplus","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.externals.six import StringIO  \nfrom IPython.display import Image  \nfrom sklearn.tree import export_graphviz\nimport pydotplus\n\ndot_data = StringIO()\nexport_graphviz(classifier, out_file=dot_data,  \n                filled=True, rounded=True,\n                special_characters=True)\ngraph = pydotplus.graph_from_dot_data(dot_data.getvalue())  \nImage(graph.create_png())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}