{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport string\nfrom collections import Counter\nimport spacy\nimport en_core_web_sm\nfrom sklearn import metrics\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn import model_selection\nfrom sklearn.metrics import classification_report\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom xgboost import XGBClassifier\nfrom sklearn import feature_extraction, linear_model, model_selection, preprocessing\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfTransformer\nimport nltk\nimport re\nfrom nltk.stem import PorterStemmer\nnlp = en_core_web_sm.load()\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\n\nnlp = en_core_web_sm.load()\nen_stop = set(stopwords.words('english'))\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-21T03:52:07.753119Z","iopub.execute_input":"2021-12-21T03:52:07.754030Z","iopub.status.idle":"2021-12-21T03:52:09.105166Z","shell.execute_reply.started":"2021-12-21T03:52:07.753982Z","shell.execute_reply":"2021-12-21T03:52:09.104455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Carregando datasets\n\ntrain = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\n\nsample_submission = pd.read_csv('../input/quora-insincere-questions-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:09.106691Z","iopub.execute_input":"2021-12-21T03:52:09.107077Z","iopub.status.idle":"2021-12-21T03:52:13.427416Z","shell.execute_reply.started":"2021-12-21T03:52:09.107048Z","shell.execute_reply":"2021-12-21T03:52:13.426591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tornar o dataset de treino balanceado.\nX_train = train[train['target'] == 1]\n\nX_train = X_train.append(train[train['target'] == 0]\n                         .sample(n = len(X_train))).reset_index(drop = True)\n\nprint(X_train.target.value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:13.428783Z","iopub.execute_input":"2021-12-21T03:52:13.429032Z","iopub.status.idle":"2021-12-21T03:52:13.652565Z","shell.execute_reply.started":"2021-12-21T03:52:13.429002Z","shell.execute_reply":"2021-12-21T03:52:13.651642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:13.653952Z","iopub.execute_input":"2021-12-21T03:52:13.654157Z","iopub.status.idle":"2021-12-21T03:52:13.663296Z","shell.execute_reply.started":"2021-12-21T03:52:13.654121Z","shell.execute_reply":"2021-12-21T03:52:13.662297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pré processamento dos dados\ndef pre(texto):\n    texto = texto.lower()\n    texto = re.sub('\\[.*?\\]', '', texto)\n    texto = re.sub('https?://\\S+|www\\.\\S+', '', texto)\n    texto = re.sub('<.*?>+', '', texto)\n    texto = re.sub('[%s]' % re.escape(string.punctuation), '', texto)\n    texto = re.sub('\\n', '', texto)\n    texto = re.sub('\\w*\\d\\w*', '', texto)\n    return texto","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:13.665576Z","iopub.execute_input":"2021-12-21T03:52:13.665793Z","iopub.status.idle":"2021-12-21T03:52:13.675581Z","shell.execute_reply.started":"2021-12-21T03:52:13.665767Z","shell.execute_reply":"2021-12-21T03:52:13.674939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Recuperando classes das instâncias\ny_target_train = X_train.target.values\n\n\n#Pré Processamento\nX_train['clean_text'] = X_train.question_text.apply(str).apply(lambda text: pre(text))\ntest['clean_text'] = test.question_text.apply(str).apply(lambda text: pre(text))\n\n\n#Obtendo os valores de treino e teste das instâncias\nX = X_train.clean_text.values\ny_test = test.clean_text.values","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:13.676831Z","iopub.execute_input":"2021-12-21T03:52:13.677114Z","iopub.status.idle":"2021-12-21T03:52:30.806453Z","shell.execute_reply.started":"2021-12-21T03:52:13.677083Z","shell.execute_reply":"2021-12-21T03:52:30.805454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Campo Clean_Text depois do pré processamento\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:30.807646Z","iopub.execute_input":"2021-12-21T03:52:30.807894Z","iopub.status.idle":"2021-12-21T03:52:30.819310Z","shell.execute_reply.started":"2021-12-21T03:52:30.807851Z","shell.execute_reply":"2021-12-21T03:52:30.818311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extração das features\nvectorizer = TfidfVectorizer(use_idf=True, stop_words='english')\n\ntfidf_model = vectorizer.fit(X)\n\ntfidf_train = tfidf_model.transform(X)\ntfidf_test = tfidf_model.transform(y_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:30.820239Z","iopub.execute_input":"2021-12-21T03:52:30.820447Z","iopub.status.idle":"2021-12-21T03:52:44.419845Z","shell.execute_reply.started":"2021-12-21T03:52:30.820421Z","shell.execute_reply":"2021-12-21T03:52:44.418858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Modelo que será avaliado é o NaiveBayes\nmodels = [\n          ('MNB', MultinomialNB())\n        ]\n\nfor name, model in models:\n        model = OneVsRestClassifier(model)\n        clf = model.fit(tfidf_train, y_target_train)\n        preds = clf.predict(tfidf_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:44.421282Z","iopub.execute_input":"2021-12-21T03:52:44.421571Z","iopub.status.idle":"2021-12-21T03:52:44.578316Z","shell.execute_reply.started":"2021-12-21T03:52:44.421532Z","shell.execute_reply":"2021-12-21T03:52:44.577397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Criando um vetor unidimensional com os qids e predições.\npreds = pd.Series(preds)\n\n#Salvando os dados finais.\nsample_submission['prediction'] = preds\nprint(sample_submission)\nsample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T03:52:44.582125Z","iopub.execute_input":"2021-12-21T03:52:44.582388Z","iopub.status.idle":"2021-12-21T03:52:45.529160Z","shell.execute_reply.started":"2021-12-21T03:52:44.582361Z","shell.execute_reply":"2021-12-21T03:52:45.528190Z"},"trusted":true},"execution_count":null,"outputs":[]}]}