{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Notebook Imports & Setup\nfrom collections import Counter, defaultdict\nfrom functools import partial\nfrom tqdm.auto import tqdm\nfrom pathlib import Path\nfrom time import time\nimport pandas as pd\nimport numpy as np\nimport sklearn\nimport joblib\nimport re\n\nfrom gensim.parsing.preprocessing import remove_stopwords\n\nfrom sklearn.linear_model import Ridge, RidgeCV, RidgeClassifier\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import SGDClassifier\nimport sklearn.pipeline\nfrom sklearn.metrics import f1_score, jaccard_score, accuracy_score, precision_score, recall_score\nfrom sklearn.metrics import mean_absolute_error, r2_score, mean_squared_error\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import SGDClassifier, Perceptron\nfrom nltk.stem import LancasterStemmer\nfrom nltk.tokenize import sent_tokenize, word_tokenize","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-16T09:40:31.257963Z","iopub.execute_input":"2022-01-16T09:40:31.258241Z","iopub.status.idle":"2022-01-16T09:40:31.269695Z","shell.execute_reply.started":"2022-01-16T09:40:31.258211Z","shell.execute_reply":"2022-01-16T09:40:31.268480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bs4 import BeautifulSoup\nlancaster=LancasterStemmer()\ndef text_cleaning(text):\n    '''\n    Cleans text into a basic form for NLP. Operations include the following:-\n    1. Remove special charecters like &, #, etc\n    2. Removes extra spaces\n    3. Removes embedded URL links\n    4. Removes HTML tags\n    5. Removes emojis\n    \n    text - Text piece to be cleaned.\n    '''\n    template = re.compile(r'https?://\\S+|www\\.\\S+') #Removes website links\n    text = template.sub(r'', text)\n    \n    soup = BeautifulSoup(text, 'lxml') #Removes HTML tags\n    only_text = soup.get_text()\n    text = only_text\n    text = re.sub(r\"[^a-zA-Z\\d]\", \" \", text) #Remove special Charecters\n    text = re.sub(' +', ' ', text) #Remove Extra Spaces\n    text = text.strip() # remove spaces at the beginning and at the end of string\n    text = remove_stopwords(text)\n    token_words=word_tokenize(text)\n    token_words\n    stem_sentence=[]\n    for word in token_words:\n        stem_sentence.append(lancaster.stem(word))\n        stem_sentence.append(\" \")\n    return \"\".join(stem_sentence)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:40:31.271617Z","iopub.execute_input":"2022-01-16T09:40:31.272017Z","iopub.status.idle":"2022-01-16T09:40:31.292334Z","shell.execute_reply.started":"2022-01-16T09:40:31.271985Z","shell.execute_reply":"2022-01-16T09:40:31.291228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_train = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/train.csv')\nold_train['y'] = 0\n#for feat, wt in FEATURE_WTS.items(): \n#    old_train.y += wt*old_train[feat]\nold_train['y'] = old_train.loc[:, 'toxic':'identity_hate'].sum(axis=1)\n#old_train.y = old_train.y/old_train.y.max()\n    \npos = old_train[old_train.y>0]\nneg = old_train[old_train.y==0].sample(len(pos)//2, random_state=201)\nold_train = pd.concat([pos, neg])\nold_train","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:40:31.293776Z","iopub.execute_input":"2022-01-16T09:40:31.294140Z","iopub.status.idle":"2022-01-16T09:40:32.788823Z","shell.execute_reply.started":"2022-01-16T09:40:31.294107Z","shell.execute_reply":"2022-01-16T09:40:32.788256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_old_test(): \n    df_test = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/test.csv')\n    df_test_labels = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/test_labels.csv')\n    df = pd.merge(df_test, df_test_labels, how='left', on = 'id')\n    df = df.replace(-1, np.nan).dropna()\n    return df\n\nold_test = read_old_test()\nold_test['y'] = 0\n#for feat, wt in FEATURE_WTS.items(): \n#    old_test.y += wt * old_test[feat]\n#old_test.y = old_test.y / old_test.y.max()\nold_test['y'] = old_test.loc[:, 'toxic':'identity_hate'].sum(axis=1)\nold_test_pos = old_test[old_test.y>0]\n\ntrain = pd.concat([old_train, old_test_pos])","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:40:32.790687Z","iopub.execute_input":"2022-01-16T09:40:32.791069Z","iopub.status.idle":"2022-01-16T09:40:34.484449Z","shell.execute_reply.started":"2022-01-16T09:40:32.791039Z","shell.execute_reply":"2022-01-16T09:40:34.483019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop('y', axis=1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:40:34.486043Z","iopub.execute_input":"2022-01-16T09:40:34.486289Z","iopub.status.idle":"2022-01-16T09:40:34.518175Z","shell.execute_reply.started":"2022-01-16T09:40:34.486254Z","shell.execute_reply":"2022-01-16T09:40:34.516351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\ntrain.comment_text = train.comment_text.progress_apply(text_cleaning)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:40:34.519961Z","iopub.execute_input":"2022-01-16T09:40:34.520202Z","iopub.status.idle":"2022-01-16T09:41:19.006315Z","shell.execute_reply.started":"2022-01-16T09:40:34.520178Z","shell.execute_reply":"2022-01-16T09:41:19.005418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn.linear_model\nimport sklearn.pipeline","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:19.007556Z","iopub.execute_input":"2022-01-16T09:41:19.007786Z","iopub.status.idle":"2022-01-16T09:41:19.011742Z","shell.execute_reply.started":"2022-01-16T09:41:19.007758Z","shell.execute_reply":"2022-01-16T09:41:19.010665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vec = TfidfVectorizer(\n        min_df=3, max_df=0.5, \n        analyzer='char_wb', ngram_range = (3,5), \n        lowercase=True, max_features=50000,\n    )","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:19.012746Z","iopub.execute_input":"2022-01-16T09:41:19.012921Z","iopub.status.idle":"2022-01-16T09:41:19.044442Z","shell.execute_reply.started":"2022-01-16T09:41:19.012899Z","shell.execute_reply":"2022-01-16T09:41:19.043270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = \\\n    sklearn.model_selection.train_test_split(train['comment_text'], train.loc[:, 'toxic':'identity_hate'],\n                                    test_size=0.20,\n                                     random_state=0\n                                    )\n\nX_train = vec.fit_transform(X_train)\nX_test = vec.transform(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:19.046656Z","iopub.execute_input":"2022-01-16T09:41:19.047058Z","iopub.status.idle":"2022-01-16T09:41:30.280216Z","shell.execute_reply.started":"2022-01-16T09:41:19.047025Z","shell.execute_reply":"2022-01-16T09:41:30.278933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_model(model, params, cv=5, scoring='f1_weighted' ):\n    model_gs = GridSearchCV(model, params, cv=cv, scoring=scoring)\n    model_gs.fit(X_train, y_train)\n    \n    y_pred = model_gs.predict(X_test)\n    y_true = y_test\n    \n    metrics_scored = [f1_score, jaccard_score, recall_score, precision_score ]\n    \n    scores = [accuracy_score(y_true,y_pred)]\n    scores += [metric(y_true, y_pred,average='weighted') for metric in metrics_scored]\n    \n    \n        \n    return model_gs, scores, y_pred","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:30.282870Z","iopub.execute_input":"2022-01-16T09:41:30.283115Z","iopub.status.idle":"2022-01-16T09:41:30.289917Z","shell.execute_reply.started":"2022-01-16T09:41:30.283092Z","shell.execute_reply":"2022-01-16T09:41:30.288296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FAST = True\nif FAST:\n    params = {\n        'estimator__penalty' : [\"l2\"],\n        'estimator__loss':['hinge'],\n        'estimator__class_weight': [None],\n        'estimator__n_jobs': [-1],\n    }\nelse:\n    params = {\n        'estimator__penalty' : [\"l1\",\"l2\",\"elasticnet\"],\n        'estimator__loss':['squared_hinge','log','hinge'],\n        'estimator__class_weight': [None,\"dict\",\"balanced\"],\n        'estimator__n_jobs': [-1],\n    }\n\nmodel = SGDClassifier()\nmodel1, scores, y_pred = fit_model(OneVsRestClassifier(model),params);\nscores, model1.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:30.291118Z","iopub.execute_input":"2022-01-16T09:41:30.291292Z","iopub.status.idle":"2022-01-16T09:41:38.287118Z","shell.execute_reply.started":"2022-01-16T09:41:30.291269Z","shell.execute_reply":"2022-01-16T09:41:38.286135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FAST = True\nif FAST:\n    params = {\n        'estimator__alpha':[4],\n    }\nelse:\n    params = {\n        'estimator__alpha':[0.5,1,1.5,2,2.5,3,3.5,4,4.5],\n    }\n\nmodel = RidgeClassifier()\n\n\nmodel2, scores, y_pred = fit_model(OneVsRestClassifier(model),params,cv=5, scoring='accuracy');\nscores, model2.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:41:38.288995Z","iopub.execute_input":"2022-01-16T09:41:38.289226Z","iopub.status.idle":"2022-01-16T09:42:03.948194Z","shell.execute_reply.started":"2022-01-16T09:41:38.289202Z","shell.execute_reply":"2022-01-16T09:42:03.947599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/jigsaw-toxic-severity-rating/comments_to_score.csv')\nsub.text = sub.text.progress_apply(text_cleaning)\nsub\n","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:42:03.949511Z","iopub.execute_input":"2022-01-16T09:42:03.949965Z","iopub.status.idle":"2022-01-16T09:42:15.587102Z","shell.execute_reply.started":"2022-01-16T09:42:03.949928Z","shell.execute_reply":"2022-01-16T09:42:15.585750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURE_WTS = {\n    'toxic': 0.32,\n    'severe_toxic': 1.5,\n    'obscene': 0.16, \n    'threat': 1.5,\n    'insult': 0.64,\n    'identity_hate': 1.5\n}\nf = np.array(list(FEATURE_WTS.values()))\nf","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:42:15.588999Z","iopub.execute_input":"2022-01-16T09:42:15.589314Z","iopub.status.idle":"2022-01-16T09:42:15.599116Z","shell.execute_reply.started":"2022-01-16T09:42:15.589277Z","shell.execute_reply":"2022-01-16T09:42:15.598039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1 = model1.decision_function(vec.transform(sub.text))\np2 = model2.decision_function(vec.transform(sub.text))\nsub['score'] = (np.array([sum(row) for row in f*p1])+np.array([sum(row) for row in f*p2]))/2\nsub","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:46:04.787242Z","iopub.execute_input":"2022-01-16T09:46:04.788017Z","iopub.status.idle":"2022-01-16T09:46:14.731255Z","shell.execute_reply.started":"2022-01-16T09:46:04.787988Z","shell.execute_reply":"2022-01-16T09:46:14.730171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[['comment_id', 'score']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-16T09:46:25.699514Z","iopub.execute_input":"2022-01-16T09:46:25.699862Z","iopub.status.idle":"2022-01-16T09:46:25.736458Z","shell.execute_reply.started":"2022-01-16T09:46:25.699832Z","shell.execute_reply":"2022-01-16T09:46:25.735766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}