{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-13T19:20:15.200226Z","iopub.execute_input":"2021-10-13T19:20:15.200578Z","iopub.status.idle":"2021-10-13T19:20:15.212567Z","shell.execute_reply.started":"2021-10-13T19:20:15.200541Z","shell.execute_reply":"2021-10-13T19:20:15.211304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/sample_submission.csv')\ntrain = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-13T19:20:15.215206Z","iopub.execute_input":"2021-10-13T19:20:15.215642Z","iopub.status.idle":"2021-10-13T19:20:22.094503Z","shell.execute_reply.started":"2021-10-13T19:20:15.215593Z","shell.execute_reply":"2021-10-13T19:20:22.093239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from unidecode import unidecode\nimport spacy\nimport re\n\nnlp = spacy.load('en_core_web_sm')\nstop_words = nlp.Defaults.stop_words","metadata":{"execution":{"iopub.status.busy":"2021-10-13T19:20:22.096217Z","iopub.execute_input":"2021-10-13T19:20:22.096584Z","iopub.status.idle":"2021-10-13T19:20:24.259569Z","shell.execute_reply.started":"2021-10-13T19:20:22.096543Z","shell.execute_reply":"2021-10-13T19:20:24.258761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pre(text):\n    sent = []\n    text = text.strip().lower()\n    text = re.sub('[0-9]{5,}','#####', text)\n    text = re.sub('[0-9]{4,}','####', text)\n    text = re.sub('[0-9]{3,}','###', text)\n    text = re.sub('[0-9]{2,}','##', text)\n    text = re.sub(r'(@.*?)[\\s]', ' ', text)\n    text = re.sub(r'([\\'\\\"\\.\\(\\)\\!\\?\\\\\\/\\,])', r' \\1 ', text)\n    text = re.sub(r'[^\\w\\s\\?]', ' ', text)\n    text = re.sub(r'([\\;\\:\\|•«\\n])', ' ', text)\n    text = re.sub('https?://\\S+|www\\.\\S+', '', text)\n    text = re.sub('<.*?>+', '', text)\n    roman = re.compile(r'^M{0,4}(CM|CD|D?C{0,3})(XC|XL|L?X{0,3})(IX|IV|V?I{0,3})$')\n    text = roman.sub(r'', text)\n    doc = nlp(text)\n    for word in doc:\n        if word.pos_ == \"VERB\":\n            sent.append(word.lemma_)\n        else:\n            sent.append(word.orth_)\n    return \" \".join(sent)","metadata":{"execution":{"iopub.status.busy":"2021-10-13T19:20:24.260739Z","iopub.execute_input":"2021-10-13T19:20:24.260976Z","iopub.status.idle":"2021-10-13T19:20:24.271336Z","shell.execute_reply.started":"2021-10-13T19:20:24.260948Z","shell.execute_reply":"2021-10-13T19:20:24.270602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"question_text\"] = train.question_text.apply(unidecode)\ntrain[\"question_text\"] = train.question_text.apply(pre)\ntest[\"question_text\"] = test.question_text.apply(unidecode)\ntest[\"question_text\"] = test.question_text.apply(pre)","metadata":{"execution":{"iopub.status.busy":"2021-10-13T19:20:24.273991Z","iopub.execute_input":"2021-10-13T19:20:24.274359Z","iopub.status.idle":"2021-10-13T23:40:10.307638Z","shell.execute_reply.started":"2021-10-13T19:20:24.274317Z","shell.execute_reply":"2021-10-13T23:40:10.30502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer, TfidfTransformer\nfrom xgboost import XGBClassifier\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.pipeline import Pipeline","metadata":{"execution":{"iopub.status.busy":"2021-10-13T23:40:10.3126Z","iopub.execute_input":"2021-10-13T23:40:10.312911Z","iopub.status.idle":"2021-10-13T23:40:11.451523Z","shell.execute_reply.started":"2021-10-13T23:40:10.312858Z","shell.execute_reply":"2021-10-13T23:40:11.450757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(ngram_range = (1, 1), use_idf = True, norm = 'l2', stop_words = stop_words)\n\nX = train.question_text.values\ny = train.target.values\n\nvectorizer.fit(X)","metadata":{"execution":{"iopub.status.busy":"2021-10-13T23:40:11.45552Z","iopub.execute_input":"2021-10-13T23:40:11.456344Z","iopub.status.idle":"2021-10-13T23:40:36.715717Z","shell.execute_reply.started":"2021-10-13T23:40:11.456303Z","shell.execute_reply":"2021-10-13T23:40:36.714601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe1 = Pipeline([('bow', TfidfVectorizer(ngram_range = (1, 1), stop_words = stop_words, max_df = 0.5, min_df = 2)),\n                ('tfid', TfidfTransformer()),\n                ('model', XGBClassifier())])\n\npipe1.fit(X, y)\n\nprediction = pipe1.predict(test.question_text)\n\ndf.target = prediction\n\ndf.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-13T23:40:36.717499Z","iopub.execute_input":"2021-10-13T23:40:36.717933Z","iopub.status.idle":"2021-10-13T23:44:16.795758Z","shell.execute_reply.started":"2021-10-13T23:40:36.71789Z","shell.execute_reply":"2021-10-13T23:44:16.79466Z"},"trusted":true},"execution_count":null,"outputs":[]}]}