{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn import model_selection\nfrom sklearn import metrics\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.model_selection import cross_validate\nfrom nltk.stem.porter import *\nfrom sklearn.metrics import f1_score\nfrom nltk.tokenize import word_tokenize\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import precision_recall_fscore_support\nfrom sklearn.metrics import plot_roc_curve\n#import warnings\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-24T04:47:55.109192Z","iopub.execute_input":"2021-07-24T04:47:55.109643Z","iopub.status.idle":"2021-07-24T04:47:56.820391Z","shell.execute_reply.started":"2021-07-24T04:47:55.109544Z","shell.execute_reply":"2021-07-24T04:47:56.819182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.linear_model import LogisticRegression\n\ndef read_data(filename):\n    \"\"\"\n    Takes in a filename returns features and targets numpy array\n    \"\"\"\n    df = pd.read_csv(filename)\n    X = np.array(df[\"question_text\"])\n    y = np.array(df[\"target\"])\n    return X, y\n\n\nfrom string import punctuation\n\n\nclass StemmerProcess(BaseEstimator, TransformerMixin):\n    def __init__(self):\n        super().__init__()\n        self.stemmer = PorterStemmer()\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X, y=None):\n        X  = [word_tokenize(string) for string in X]\n        X = self.remove_punctuation_list(X)\n        X = [[self.stemmer.stem(word) for word in string] for string in X]\n        X = [\" \".join(token for token in list) for list in X]\n        return X\n    \n    def remove_punctuation_list(self, X):\n        return [[remove_punctuation(token) for token in string if len(token)>1] for string in X]\n\ndef remove_punctuation(word):\n    return ''.join(c for c in word if c not in punctuation)\n    \n\nclass BasicPreProcessing(BaseEstimator, TransformerMixin):\n    def __init__(self):\n        super().__init__()\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X, y=None):\n        X  = [word_tokenize(string) for string in X]\n        X = self.remove_punctuation_list(X)\n        X = [\" \".join(token for token in list) for list in X]\n        return X\n\n    def remove_punctuation_list(self, X):\n        return [[remove_punctuation(token) for token in string if len(token)>1] for string in X]\n        \n\n#add other methods as desired\n#Class inspired by https://stackoverflow.com/questions/50285973/pipeline-multiple-classifiers\nclass modelSwitcher(BaseEstimator):\n    def __init__(self, estimator=LogisticRegression()):\n        self.estimator = estimator\n\n    def fit(self, X_train, y_train=None, **kwargs):\n        self.estimator.fit(X_train,y_train)\n        return self\n\n    def predict(self, X_test, y=None):\n        return self.estimator.predict(X_test)\n\n    def predict_proba(self, X_test):\n        return self.estimator.predict_proba(X_test)\n\n    def score(self, X, y):\n        return self.estimator.score(X, y)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:47:58.407301Z","iopub.execute_input":"2021-07-24T04:47:58.40768Z","iopub.status.idle":"2021-07-24T04:47:58.426046Z","shell.execute_reply.started":"2021-07-24T04:47:58.407647Z","shell.execute_reply":"2021-07-24T04:47:58.424952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X,y = read_data(\"../input/quora-insincere-questions-classification/train.csv\")\nX_train, X_test, y_train, y_test = train_test_split(X,y, test_size = 0.25, random_state=0)\nprint(len(X_train), len(X_test))","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:48:04.788942Z","iopub.execute_input":"2021-07-24T04:48:04.789529Z","iopub.status.idle":"2021-07-24T04:48:09.686903Z","shell.execute_reply.started":"2021-07-24T04:48:04.789492Z","shell.execute_reply":"2021-07-24T04:48:09.685711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Best_LR = make_pipeline(StemmerProcess(), CountVectorizer(ngram_range=(1,3)), LogisticRegression(C=1, solver=\"liblinear\", class_weight=\"balanced\",penalty=\"l2\", max_iter=200, verbose=1, n_jobs=-1))\nBest_MNB = make_pipeline(BasicPreProcessing(), CountVectorizer(ngram_range=(1,3)), MultinomialNB(alpha=0.1))\nBest_SVM = make_pipeline(StemmerProcess(), TfidfVectorizer(ngram_range=(1,2)), LinearSVC(C=1, penalty = \"l2\", fit_intercept=False, dual=False, loss=\"squared_hinge\", class_weight=\"balanced\", max_iter=2000, verbose=1))\nBest_Forest = make_pipeline(StemmerProcess(), CountVectorizer(ngram_range=(1,1)), RandomForestClassifier(min_samples_split = int(len(X_train)*0.01),n_estimators=500, max_features=\"log2\", class_weight=\"balanced\", verbose=1, n_jobs=-1))\nBest_models = [Best_LR, Best_MNB, Best_SVM, Best_Forest]","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:48:09.688835Z","iopub.execute_input":"2021-07-24T04:48:09.689329Z","iopub.status.idle":"2021-07-24T04:48:09.699933Z","shell.execute_reply.started":"2021-07-24T04:48:09.68928Z","shell.execute_reply":"2021-07-24T04:48:09.698604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Names = [\"Logistic Regression\", \"Multinomial Naive Bayes\", \"Support Vector Machine\", \"Random Forest\"]\nfor model in Best_models:\n    model.fit(X_train, y_train)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:48:21.085815Z","iopub.execute_input":"2021-07-24T04:48:21.086228Z","iopub.status.idle":"2021-07-24T04:53:48.298556Z","shell.execute_reply.started":"2021-07-24T04:48:21.086195Z","shell.execute_reply":"2021-07-24T04:53:48.297366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:25:06.995346Z","iopub.execute_input":"2021-07-24T04:25:06.99572Z","iopub.status.idle":"2021-07-24T04:26:29.577724Z","shell.execute_reply.started":"2021-07-24T04:25:06.995679Z","shell.execute_reply":"2021-07-24T04:26:29.576767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ensemble_predictions(predictions):\n    #predictions is an 4 x len(train) array \n    preds = []\n    for i in range(len(predictions[0])):\n        pred = 0\n        for model in range(4):\n            if model == 0: #logistic regression\n                pred += predictions[model][i] *1.3\n            elif model == 4: #random forest\n                pred += predictions[model][i] *0.5\n            else: #svm or MNB\n                pred += predictions[model][i] *1.10\n        if pred / 4 > 0.55: #threshold for class classification\n            preds.append(1)\n        else:\n            preds.append(0)\n    return preds\n#ensemble_preds = ensemble_predictions(predictions_all_models)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:53:48.300409Z","iopub.execute_input":"2021-07-24T04:53:48.300799Z","iopub.status.idle":"2021-07-24T04:53:48.308635Z","shell.execute_reply.started":"2021-07-24T04:53:48.300765Z","shell.execute_reply":"2021-07-24T04:53:48.307356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\nX_submission = np.array(df[\"question_text\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:53:48.310856Z","iopub.execute_input":"2021-07-24T04:53:48.311337Z","iopub.status.idle":"2021-07-24T04:53:49.755913Z","shell.execute_reply.started":"2021-07-24T04:53:48.311286Z","shell.execute_reply":"2021-07-24T04:53:49.754825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ensumble model submissions\npredictions_submissions = []\nfor model in Best_models:\n    predictions_submissions.append(model.predict(X_submission))\nensemble_submissions = ensemble_predictions(predictions_submissions)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:54:56.436639Z","iopub.execute_input":"2021-07-24T04:54:56.437086Z","iopub.status.idle":"2021-07-24T05:10:16.546179Z","shell.execute_reply.started":"2021-07-24T04:54:56.437045Z","shell.execute_reply":"2021-07-24T05:10:16.544958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame.from_dict({'qid' : df['qid']})\nsubmission['prediction'] = ensemble_submissions\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T05:10:50.257834Z","iopub.execute_input":"2021-07-24T05:10:50.258249Z","iopub.status.idle":"2021-07-24T05:10:51.352632Z","shell.execute_reply.started":"2021-07-24T05:10:50.258218Z","shell.execute_reply":"2021-07-24T05:10:51.351527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}