{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic imports\n \nimport torch \nimport numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-17T17:51:21.832005Z","iopub.execute_input":"2022-11-17T17:51:21.832487Z","iopub.status.idle":"2022-11-17T17:51:21.838478Z","shell.execute_reply.started":"2022-11-17T17:51:21.832448Z","shell.execute_reply":"2022-11-17T17:51:21.837255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import the data \ndata = pd.read_csv(\"../input/predict-closed-questions-on-stack-overflow/train-sample.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:21.846413Z","iopub.execute_input":"2022-11-17T17:51:21.846834Z","iopub.status.idle":"2022-11-17T17:51:24.086526Z","shell.execute_reply.started":"2022-11-17T17:51:21.846798Z","shell.execute_reply":"2022-11-17T17:51:24.085391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.OpenStatus.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:24.088327Z","iopub.execute_input":"2022-11-17T17:51:24.088659Z","iopub.status.idle":"2022-11-17T17:51:24.107899Z","shell.execute_reply.started":"2022-11-17T17:51:24.088629Z","shell.execute_reply":"2022-11-17T17:51:24.106825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's take 'TITLE' & 'BODYMARKDOWN' & OpenStatus Columns \ndata_train = data[['Title', 'BodyMarkdown', 'OpenStatus']]\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:24.109943Z","iopub.execute_input":"2022-11-17T17:51:24.110390Z","iopub.status.idle":"2022-11-17T17:51:24.149271Z","shell.execute_reply.started":"2022-11-17T17:51:24.110348Z","shell.execute_reply":"2022-11-17T17:51:24.147919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.shape  # it's tooo big, let's take just 20 k rows. ","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:30.918112Z","iopub.execute_input":"2022-11-17T17:51:30.918634Z","iopub.status.idle":"2022-11-17T17:51:30.928162Z","shell.execute_reply.started":"2022-11-17T17:51:30.918588Z","shell.execute_reply":"2022-11-17T17:51:30.926692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample 80k randomly! \ndata_train = data_train.sample(80000, random_state = 234)\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:31.094303Z","iopub.execute_input":"2022-11-17T17:51:31.094942Z","iopub.status.idle":"2022-11-17T17:51:31.127816Z","shell.execute_reply.started":"2022-11-17T17:51:31.094894Z","shell.execute_reply":"2022-11-17T17:51:31.126571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install texthero  # For text pre processing ","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:31.263317Z","iopub.execute_input":"2022-11-17T17:51:31.263730Z","iopub.status.idle":"2022-11-17T17:51:42.362955Z","shell.execute_reply.started":"2022-11-17T17:51:31.263694Z","shell.execute_reply":"2022-11-17T17:51:42.361366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's pre process the data using library called \"Text_hero\"\nimport texthero as hero \n\ndata_train['Title'] = hero.clean(data_train['Title'])\ndata_train['BodyMarkdown'] = hero.clean(data_train['BodyMarkdown'])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:42.365517Z","iopub.execute_input":"2022-11-17T17:51:42.365904Z","iopub.status.idle":"2022-11-17T17:52:01.848025Z","shell.execute_reply.started":"2022-11-17T17:51:42.365850Z","shell.execute_reply":"2022-11-17T17:52:01.846878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"```Python \nhero.clear(df['something'])  \n\n# It does many things, They are: \nfillna()\nlowercase()\nremove_digits()\nremove_punctuation()\nremove_diacritics()\nremove_stopwords()\nremove_whitespace()\n```","metadata":{}},{"cell_type":"code","source":"# Let's see the data now! \ndata_train.head()  # look how clean our text is:) ","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:01.849488Z","iopub.execute_input":"2022-11-17T17:52:01.849926Z","iopub.status.idle":"2022-11-17T17:52:01.861840Z","shell.execute_reply.started":"2022-11-17T17:52:01.849895Z","shell.execute_reply":"2022-11-17T17:52:01.860630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's change the target to numbers: \ndata_train.OpenStatus.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:01.864373Z","iopub.execute_input":"2022-11-17T17:52:01.864754Z","iopub.status.idle":"2022-11-17T17:52:01.880437Z","shell.execute_reply.started":"2022-11-17T17:52:01.864703Z","shell.execute_reply":"2022-11-17T17:52:01.879208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['OpenStatus']= data_train['OpenStatus'].map({'open': 0, 'not a real question': 1, 'off topic': 2, 'not constructive': 3, 'too localized': 4}) \ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:01.881829Z","iopub.execute_input":"2022-11-17T17:52:01.882912Z","iopub.status.idle":"2022-11-17T17:52:02.343209Z","shell.execute_reply.started":"2022-11-17T17:52:01.882827Z","shell.execute_reply":"2022-11-17T17:52:02.342309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's seperate x and y: \nx = data_train['Title'] + ' '+ data_train['BodyMarkdown']\ny = data_train['OpenStatus']\n\n# xtrain, ytrian \nfrom sklearn.model_selection import train_test_split \nxtrain, xtest, ytrain, ytest = train_test_split(x, y, test_size = 0.3, random_state = 203)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:02.344687Z","iopub.execute_input":"2022-11-17T17:52:02.345389Z","iopub.status.idle":"2022-11-17T17:52:02.416348Z","shell.execute_reply.started":"2022-11-17T17:52:02.345356Z","shell.execute_reply":"2022-11-17T17:52:02.415253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"```Python \nLet us go from basic to Advance modeling :) \n```","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:18:42.995019Z","iopub.execute_input":"2022-07-05T04:18:42.995416Z","iopub.status.idle":"2022-07-05T04:18:43.004567Z","shell.execute_reply.started":"2022-07-05T04:18:42.995386Z","shell.execute_reply":"2022-07-05T04:18:43.003419Z"}}},{"cell_type":"code","source":"# Let's covvert words to numbers using TF-IDF \nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nvectorizer = TfidfVectorizer(max_features = 10000)  # it contains only 10k features (fixed!)\n\nxtrain_tfidf = vectorizer.fit_transform(xtrain).toarray()  # converting words to numbers for train data \nxtest_tfidf = vectorizer.transform(xtest).toarray()        # converting words to numbers for test data ","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:02.418192Z","iopub.execute_input":"2022-11-17T17:52:02.419180Z","iopub.status.idle":"2022-11-17T17:52:15.200746Z","shell.execute_reply.started":"2022-11-17T17:52:02.419146Z","shell.execute_reply":"2022-11-17T17:52:15.199666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\npickle.dump(vectorizer, open(\"vectorizer.pickle\", \"wb\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:55:19.896535Z","iopub.execute_input":"2022-11-17T17:55:19.897049Z","iopub.status.idle":"2022-11-17T17:55:19.990633Z","shell.execute_reply.started":"2022-11-17T17:55:19.897011Z","shell.execute_reply":"2022-11-17T17:55:19.989744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stop Running","metadata":{}},{"cell_type":"code","source":"# Naive Bayes \nfrom sklearn.naive_bayes import GaussianNB\nclf = GaussianNB()\n\nclf.fit(xtrain_tfidf, ytrain)\n\npredicted_naive = clf.predict(xtest_tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:55:23.174291Z","iopub.execute_input":"2022-11-17T17:55:23.174668Z","iopub.status.idle":"2022-11-17T17:55:38.702101Z","shell.execute_reply.started":"2022-11-17T17:55:23.174637Z","shell.execute_reply":"2022-11-17T17:55:38.700761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics :) \nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix \n\nprint('Accuracy Score \\n',accuracy_score(predicted_naive, ytest))\nprint('Confusion Matrix \\n', confusion_matrix(predicted_naive, ytest))\nprint('Classification Report \\n', classification_report(predicted_naive, ytest))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:55:38.704481Z","iopub.execute_input":"2022-11-17T17:55:38.705022Z","iopub.status.idle":"2022-11-17T17:55:38.767590Z","shell.execute_reply.started":"2022-11-17T17:55:38.704967Z","shell.execute_reply":"2022-11-17T17:55:38.766361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfilename = 'g_naive.sav'\npickle.dump(clf, open(filename, 'wb'))\n \n# some time later...","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:55:38.769288Z","iopub.execute_input":"2022-11-17T17:55:38.769734Z","iopub.status.idle":"2022-11-17T17:55:38.776762Z","shell.execute_reply.started":"2022-11-17T17:55:38.769692Z","shell.execute_reply":"2022-11-17T17:55:38.775577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model from disk\nloaded_model = pickle.load(open(filename, 'rb'))\nnaive = loaded_model.predict(xtest_tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:15.235397Z","iopub.status.idle":"2022-11-17T17:52:15.235753Z","shell.execute_reply.started":"2022-11-17T17:52:15.235570Z","shell.execute_reply":"2022-11-17T17:52:15.235587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Start now","metadata":{}},{"cell_type":"code","source":"# MLP classifier \nfrom sklearn.neural_network import MLPClassifier\n\nmlp_cv=MLPClassifier(early_stopping=True, verbose=2)\nmlp_cv.fit(xtrain_tfidf, ytrain)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:55:40.665853Z","iopub.execute_input":"2022-11-17T17:55:40.666264Z","iopub.status.idle":"2022-11-17T17:58:51.872803Z","shell.execute_reply.started":"2022-11-17T17:55:40.666232Z","shell.execute_reply":"2022-11-17T17:58:51.871563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict :) \npredicted_mlp = mlp_cv.predict(xtest_tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:58:51.875221Z","iopub.execute_input":"2022-11-17T17:58:51.875972Z","iopub.status.idle":"2022-11-17T17:58:53.223531Z","shell.execute_reply.started":"2022-11-17T17:58:51.875928Z","shell.execute_reply":"2022-11-17T17:58:53.221920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics :) \nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix \n\ndef metrics(predicted): \n    predicted_naive = predicted \n    print('Accuracy Score \\n',accuracy_score(predicted_naive, ytest))\n    print('Confusion Matrix \\n', confusion_matrix(predicted_naive, ytest))\n    print('Classification Report \\n', classification_report(predicted_naive, ytest))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:15.243430Z","iopub.status.idle":"2022-11-17T17:52:15.244030Z","shell.execute_reply.started":"2022-11-17T17:52:15.243822Z","shell.execute_reply":"2022-11-17T17:52:15.243842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics(predicted_mlp)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:52:15.245206Z","iopub.status.idle":"2022-11-17T17:52:15.245557Z","shell.execute_reply.started":"2022-11-17T17:52:15.245381Z","shell.execute_reply":"2022-11-17T17:52:15.245398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfilename = 'mlp.sav'\npickle.dump(mlp_cv, open(filename, 'wb'))\n \n# some time later...\n \n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:58:53.226083Z","iopub.execute_input":"2022-11-17T17:58:53.226938Z","iopub.status.idle":"2022-11-17T17:58:53.276310Z","shell.execute_reply.started":"2022-11-17T17:58:53.226887Z","shell.execute_reply":"2022-11-17T17:58:53.274635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model from disk\nloaded_model = pickle.load(open(filename, 'rb'))\npredicted_mlp2 = loaded_model.predict(xtest_tfidf)\nmetrics(predicted_mlp2)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:51:24.249549Z","iopub.status.idle":"2022-11-17T17:51:24.249989Z","shell.execute_reply.started":"2022-11-17T17:51:24.249757Z","shell.execute_reply":"2022-11-17T17:51:24.249777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.datasets import make_moons, make_circles, make_classification\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.gaussian_process import GaussianProcessClassifier\nfrom sklearn.gaussian_process.kernels import RBF\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:59:05.744025Z","iopub.execute_input":"2022-11-17T17:59:05.744448Z","iopub.status.idle":"2022-11-17T17:59:05.752286Z","shell.execute_reply.started":"2022-11-17T17:59:05.744403Z","shell.execute_reply":"2022-11-17T17:59:05.751096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MLP classifier \n\nsvc=SVC(kernel=\"linear\", C=1)\nsvc.fit(xtrain_tfidf, ytrain)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:59:16.439622Z","iopub.execute_input":"2022-11-17T17:59:16.440010Z","iopub.status.idle":"2022-11-18T00:25:22.254940Z","shell.execute_reply.started":"2022-11-17T17:59:16.439978Z","shell.execute_reply":"2022-11-18T00:25:22.250400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = 'svc.sav'\npickle.dump(svc, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:25:22.258937Z","iopub.status.idle":"2022-11-18T00:25:22.259737Z","shell.execute_reply.started":"2022-11-18T00:25:22.259492Z","shell.execute_reply":"2022-11-18T00:25:22.259519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc=RandomForestClassifier(max_depth=5, n_estimators=10, max_features=1)\nsvc.fit(xtrain_tfidf, ytrain)\nfilename = 'rand_for.sav'\npickle.dump(svc, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:25:22.261310Z","iopub.status.idle":"2022-11-18T00:25:22.261785Z","shell.execute_reply.started":"2022-11-18T00:25:22.261576Z","shell.execute_reply":"2022-11-18T00:25:22.261599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc=QuadraticDiscriminantAnalysis()\nsvc.fit(xtrain_tfidf, ytrain)\nfilename = 'qda.sav'\npickle.dump(svc, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-11-18T00:25:22.263853Z","iopub.status.idle":"2022-11-18T00:25:22.264300Z","shell.execute_reply.started":"2022-11-18T00:25:22.264092Z","shell.execute_reply":"2022-11-18T00:25:22.264113Z"},"trusted":true},"execution_count":null,"outputs":[]}]}