{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Quora Insincere Questions Classification"},{"metadata":{},"cell_type":"markdown","source":"This is quick example using Uber's Ludwig library. A code-free way to implement and train deep learning models using state-of-the-art NN architectures."},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install https://github.com/dimension23/ludwig/archive/master.zip","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport logging\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score, roc_curve, precision_recall_curve\nfrom ludwig.api import LudwigModel\n\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_definition = {\n    \"input_features\": [\n        {\n            \"bidirectional\": True,\n            \"cell_type\": \"lstm_cudnn\",\n            \"dropout\": True,\n            \"embedding_size\": 300,\n            \"embeddings_trainable\": True,\n            \"encoder\": \"rnn\",\n            \"level\": \"word\",\n            \"name\": \"question_text\",\n            \"pretrained_embeddings\": \"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\",\n            \"type\": \"text\"\n        }\n    ],\n    \"output_features\": [\n        {\n            \"name\": \"target\",\n            \"type\": \"category\"\n        }\n    ],\n    \"preprocessing\" : {\n        \"stratify\": \"target\",\n        \"text\": {\n            \"lowercase\": True\n        }\n    }\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = LudwigModel(model_definition)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_dataframe = pd.read_csv(\"../input/train.csv\")\n\ntraining_dataframe, validation_dataframe = train_test_split(input_dataframe,\n                                                      test_size=0.1, \n                                                      random_state=42, \n                                                      stratify=input_dataframe[\"target\"])\n\ntraining_dataframe.reset_index(inplace=True)\nvalidation_dataframe.reset_index(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_stats = model.train(training_dataframe, logging_level=logging.INFO)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_stats","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions_dataframe = model.predict(validation_dataframe, logging_level=logging.INFO)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_dataframe = validation_dataframe.merge(predictions_dataframe, left_index=True, right_index=True)\nresults_dataframe[\"target_predictions\"] = pd.to_numeric(results_dataframe[\"target_predictions\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f1_score(results_dataframe[\"target\"], results_dataframe[\"target_predictions\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dataframe = pd.read_csv(\"../input/test.csv\")\ntest_predictions = model.predict(test_dataframe, logging_level=logging.INFO)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.close()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_dataframe = test_dataframe.merge(test_predictions, left_index=True, right_index=True)[[\"qid\", \"target_predictions\"]]\nsubmission_dataframe.columns = [\"qid\", \"prediction\"]\nsubmission_dataframe.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}