{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"from fastai.text import *\nfrom fastai.datasets import URLs\nimport torch\nimport pandas as pd\nimport numpy as np\nimport logging\nimport os\nimport random\nlogging.basicConfig(filename='log.txt',level=logging.DEBUG, format='%(asctime)s %(message)s')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"71a9dcb099b867b7dfa250472119943c36ba1c1a","trusted":true},"cell_type":"code","source":"!( head -535000 ../input/train.csv ) > train.csv # reducing size to train in less than 2 hours\n# !( head -200 ../input/test.csv ) > test.csv\n# !cp ../input/train.csv train.csv\n!cp ../input/test.csv test.csv\ntrain_file = 'train.csv'\ntest_file = 'test.csv'\nfolder = '.'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26cd3ac89d550bc5b2dde03cf8c95db47d5053e7","trusted":true},"cell_type":"code","source":"# Language model data (10 min with whole training set)\ndata_lm = TextLMDataBunch.from_csv(folder, \n                                   train_file, \n                                   text_cols='question_text', \n                                   label_cols='target')\nassert data_lm.device == torch.device('cuda')\nlogging.info('Language model data finish')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26cd3ac89d550bc5b2dde03cf8c95db47d5053e7","trusted":true},"cell_type":"code","source":"# Language model trainer (66 min with whole training set)\nlearn = language_model_learner(data_lm, \n                           drop_mult=0.5,\n                           bptt=45)\nlearn.fit_one_cycle(1 , 1e-2)\nlearn.save_encoder('encoder')\nlogging.info('Language model trained')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26cd3ac89d550bc5b2dde03cf8c95db47d5053e7","trusted":true},"cell_type":"code","source":"# Classifier model data (10 min with whole training set)\ndata_clas = TextClasDataBunch.from_csv(folder, \n                                       train_file,\n                                       valid_pct=0.1, #added to see if we prevent errors \n                                       vocab=data_lm.train_ds.vocab, \n                                       bs=32,\n                                       text_cols='question_text', \n                                       label_cols='target')\nlogging.info('data_clas created')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26cd3ac89d550bc5b2dde03cf8c95db47d5053e7","trusted":true},"cell_type":"code","source":"# Classifier training (43 min with whole training set)\nlearn_classifier = text_classifier_learner(data_clas, drop_mult=0.5)\nlearn_classifier.load_encoder('encoder')\nlearn_classifier.fit_one_cycle(1, 1e-2)   \nlogging.info('learn_classifier fit')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26cd3ac89d550bc5b2dde03cf8c95db47d5053e7","trusted":true},"cell_type":"code","source":"# Do predictions and submission (67 min with whole test set)\n# This is awful in terms of time and can be reduced to a few seconds (see next TODO)\n# TODO: take the model out using learn_classifier.model and do steps manually. Work in progress.\ntest_set =  pd.read_csv(test_file)\ntrue_threshold = 0.33\npredictions = test_set['question_text'].apply(lambda x: int(learn_classifier.predict(x)[2][1]>true_threshold))\nlogging.info('predictions built')\nsubmission = pd.DataFrame(test_set['qid'])\nsubmission['prediction'] = predictions \nsubmission.to_csv('submission.csv',index=False)\nlogging.info('submission built')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}