{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nfrom sklearn.model_selection import GridSearchCV\nfrom keras.preprocessing.text import Tokenizer\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.models import Sequential\nfrom keras.layers import Embedding, LSTM, Dropout, Dense, Flatten\nfrom keras.wrappers.scikit_learn import KerasClassifier\nfrom sklearn import metrics\n\n# print(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')\nsubmission_df = pd.read_csv('../input/sample_submission.csv')\nprint(train_df.columns, test_df.columns, submission_df.columns)\nprint(train_df.shape, test_df.shape, submission_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73fbda7e7c3208882aa761b3767e844531b6dada"},"cell_type":"code","source":"max_features = 5000\ntokenizer = Tokenizer(num_words= max_features)\nx_train, x_val, y_train, y_val = train_test_split(train_df.question_text, \n                                                  train_df.target, \n                                                  test_size=0.33, \n                                                  random_state=6122018, \n                                                  stratify = train_df.target)\ntokenizer.fit_on_texts(x_train)\nmaxlen = 50 # max words in a question\nx_train = tokenizer.texts_to_sequences(x_train)\nx_train = pad_sequences(x_train, maxlen=maxlen)\n\nx_val = tokenizer.texts_to_sequences(x_val)\nx_val = pad_sequences(x_val, maxlen=maxlen)\n\nx_test = tokenizer.texts_to_sequences(test_df.question_text)\nx_test = pad_sequences(x_test, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8faefc641427353e214c55519417418e248595f"},"cell_type":"code","source":"print(x_train.shape, x_val.shape, y_train.shape, y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d81a4ec3ac35dd844b5266a653ff78987777b543"},"cell_type":"code","source":"batch_size = 1024\nhidden_size = 32\nuse_dropout = True\ndef create_model(hidden_size, dropout_rate):\n    model = Sequential()\n    model.add(Embedding(max_features, 128, input_length=maxlen))\n    model.add(LSTM(hidden_size, return_sequences=True))\n    model.add(Dropout(dropout_rate))\n    model.add(LSTM(int(hidden_size/2), return_sequences=True))\n    model.add(Flatten())\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])\n    return model\n\nmodel = KerasClassifier(build_fn=create_model, epochs=2, batch_size=batch_size, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66c6770d90a3a2a674e76c5ceffc3b230004d915"},"cell_type":"code","source":"# del(train_df)\n# del(test_df)\n# del(submission_df)\n%env JOBLIB_TEMP_FOLDER=/tmp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3994028917829516ac87efbdbfeea312935dd517"},"cell_type":"code","source":"# define the grid search parameters\nhidden_size = [32]\ndropout_rate = [0.5]\n# dropout_rate = [0.0, 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9]\nparam_grid = dict(hidden_size=hidden_size, dropout_rate=dropout_rate)\ngrid = GridSearchCV(estimator=model, param_grid=param_grid, n_jobs=-1)\ngrid_result = grid.fit(x_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f82da3c3232a3b4c23dbd7bc93a1068564a1954"},"cell_type":"code","source":"# summarize results\nprint(\"Best: %f using %s\" % (grid_result.best_score_, grid_result.best_params_))\nmeans = grid_result.cv_results_['mean_test_score']\nstds = grid_result.cv_results_['std_test_score']\nparams = grid_result.cv_results_['params']\nfor mean, stdev, param in zip(means, stds, params):\n    print(\"%f (%f) with: %r\" % (mean, stdev, param))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e05c5def36e8c095c8ef982806c78c6d6cf6d262"},"cell_type":"code","source":"# help(grid_result.predict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f63ea6722987cbc5085414fd54eb6dfdcd109de1"},"cell_type":"code","source":"# pred_noemb_val_y = grid.predict([x_val], batch_size=1024, verbose=1)\npred_noemb_val_y = grid.predict([x_val])\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(y_val, (pred_noemb_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88ddf544b0c22ba8b1f83e849106720d87bd6c7c"},"cell_type":"code","source":"pred_noemb_val_y","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f9c4efa8ed76c5a5b3f9b19e525af36604f088cf","trusted":true},"cell_type":"code","source":"# push these two statemets tothe begining of the notebook\nx_test = tokenizer.texts_to_sequences(test_df.question_text)\nx_test = pad_sequences(x_test, maxlen=maxlen)\n\n# pred_noemb_val_y = model.predict([x_test], batch_size=1024, verbose=1)\npred_noemb_val_y = grid.predict([x_test])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59fe53196e96d96ae54c3ffa734f74aef8f1fd43"},"cell_type":"code","source":"import seaborn as sns\nsns.distplot(pred_noemb_val_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40e358cd038b518b44059b8e2d822b460e4be76a"},"cell_type":"code","source":"# threshold = 0.29\n# submission_df.prediction = (pred_noemb_val_y[:,0] > threshold).astype(np.int)\nsubmission_df.prediction = pred_noemb_val_y\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5877a6b1dbf3905f4aaaa7b45f4d88624f00ea7c","trusted":true,"scrolled":false},"cell_type":"code","source":"submission_df.prediction.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30d90fb36d8d20f7a5edb0aef3b07262ce3e1ca7"},"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}